Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
24 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion chemotools/models/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,8 +3,9 @@
import warnings

from chemotools.models._cross_decomposition import PLSRegression
from ._principal_component_regression import PrincipalComponentRegression

__all__ = ["PLSRegression"]
__all__ = ["PLSRegression", "PrincipalComponentRegression"]

# Show deprecation notice on module import
warnings.warn(
Expand Down
202 changes: 202 additions & 0 deletions chemotools/models/_principal_component_regression.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,202 @@
"""
The :mod:`chemotools.models._principal_component_regression` module implements a PCR model.
"""

# Authors: Ruggero Guerrini
# License: MIT

from sklearn.base import BaseEstimator, TransformerMixin, RegressorMixin
from sklearn.decomposition import PCA
from sklearn.linear_model import LinearRegression
from sklearn.utils.validation import validate_data, check_is_fitted
from numbers import Integral
from sklearn.utils._param_validation import Interval, RealNotInt, StrOptions
import numpy as np


class PrincipalComponentRegression(TransformerMixin, RegressorMixin, BaseEstimator):
"""
Implement Principal Component regression

Parameters
----------
n_components : int, default = 2
The number of components used to calculate the PCA model
# add comments on parameter constraints

Comment on lines +18 to +26

Copilot AI Apr 13, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The class docstring has several issues that will surface in generated docs: (1) it states n_components default is 2 but __init__ defaults to None; (2) the example import uses chemotools.decomposition which does not exist in this repo (should be chemotools.models); and (3) it contains a leftover placeholder line # add comments on parameter constraints. Please align the docstring with the actual public API and remove the placeholder.

Copilot uses AI. Check for mistakes.
Attributes
----------
pca_ : PCA objects from sklearn
Fitted PCA model

lr_ : Linear Regression objects from sklearn
Fitted Linear Regression model

components_ : ndarray of shape (n_components, n_features)
Loadings of the PCA model

explained_variance_ : ndarray of shape (n_components, )
Explained variance of the PCA model

explained_variance_ratio_ : ndarray of shape (n_components, )
Explained variance ratio of the PCA model

noise_variance_ : float
Equal to the average of (min(n_features, n_samples) - n_components)
smallest eigenvalues of the covariance matrix of X.

coef_ : ndarray of shape (n_components, ) or (n_targets, n_components)
Coefficients of the Linear Regression model

intercept_ : ndarray of shape (n_targets, )
Intercetps of the Linear Regression

rank_ : int
Rank from the Linear Regressio model

singular_ : ndarray of shape (min(X, y),)
Singular values of X. Only available when X is dense.

Examples
--------
>>> from chemotools.decomposition import PrincipalComponentRegression
>>> # Generate sample data
>>> X = np.random.randn(100, 50)
>>> X_test = np.random.randn(10, 50)
>>> y = X[:, 0] + 2*X[:, 1] + np.random.randn(100)*0.1
>>> # Fit model
>>> pcr = PrincipalComponentRegression(n_components=2)
>>> pcr.fit(X, y)
>>> y_hat = pcr.predict(X_test)
"""

_parameter_constraints: dict = {
"n_components": [
Interval(Integral, 0, None, closed="left"),
Interval(RealNotInt, 0, 1, closed="neither"),
StrOptions({"mle"}),
None,
],
"copy": ["boolean"],
}

def __init__(
self,
n_components: int | None = None,
copy: bool = True,
):
self.n_components = n_components
self.copy = copy
Comment on lines +73 to +89

Copilot AI Apr 13, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The parameter constraints indicate n_components may be an int, a float in (0, 1), the string "mle", or None, but the type hint in __init__ restricts it to int | None. This mismatch will confuse users/type-checkers and doesn't reflect the actual accepted values passed through to sklearn.decomposition.PCA. Consider widening the annotation (and updating the docstring) to match the supported types.

Copilot uses AI. Check for mistakes.

def __sklearn_tags__(self):
tags = super().__sklearn_tags__()
tags.target_tags.multi_output = True
return tags

def fit(self, X: np.ndarray, y: np.ndarray) -> "PrincipalComponentRegression":
"""
Fit the model to the input data.

Parameters
----------
X : np.ndarray of shape (n_samples, n_features)
The input data to fit the transformer to.

y : np.ndarray of shape (n_samples, ) or (n_samples, n_targets)
The properties to be predicted.

Returns
-------
self : PrincipalComponentRegression
The regression model.
"""
# Validate input data
self._validate_params()
X, y = validate_data(
self,
X,
y,
ensure_2d=True,
reset=True,
copy=self.copy,
dtype=np.float64,
multi_output=True,
)

# Train PCA model
pca = PCA(n_components=self.n_components).fit(X)
x_scores = pca.transform(X)

# Train linear regression model
lr = LinearRegression().fit(x_scores, y)

Comment on lines +126 to +132

Copilot AI Apr 13, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The estimator exposes a copy parameter, but it is only used in validate_data and is not passed through to the underlying PCA(copy=...) (nor documented). This makes copy behavior differ from sklearn's PCA(copy=...) and from what users might expect in a PCR wrapper. Consider either propagating copy to PCA (and possibly LinearRegression) or renaming/removing it to avoid implying sklearn-PCA semantics.

Copilot uses AI. Check for mistakes.
# Expose fitting attributes for PCA
self.mean_ = pca.mean_
self.components_ = pca.components_
self.explained_variance_ = pca.explained_variance_
self.explained_variance_ratio_ = pca.explained_variance_ratio_
self.noise_variance_ = pca.noise_variance_

# Expose fitting attributes for linear regression
self.coef_ = lr.coef_
self.intercept_ = lr.intercept_
self.rank_ = lr.rank_
self.singular_ = lr.singular_

return self

def transform(self, X: np.ndarray) -> np.ndarray:
"""
Transform new data into the PCA trained latent space

Parameters
----------
X : np.ndarray of shape (n_samples, n_features)
The input data to transform.

Returns
-------
x_scores : np.ndarray of shape (n_samples,n_components)
The transformed data.
"""

Copilot AI Apr 13, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

transform() calls self.pca_.transform(X) without verifying the estimator is fitted or validating feature count/dtype. This will raise an AttributeError instead of a scikit-learn NotFittedError, and skips validate_data(..., reset=False) checks that predict() already performs. Add check_is_fitted(self, ["pca_", "lr_"]) (or at least "pca_") and validate X similarly to predict() before transforming.

Suggested change
"""
"""
check_is_fitted(self, ["pca_", "lr_"])
X = validate_data(
self,
X,
ensure_2d=True,
reset=False,
copy=self.copy,
dtype=np.float64,
)

Copilot uses AI. Check for mistakes.
# Validate input data
check_is_fitted(self, ["mean_", "components_", "coef_", "intercept_"])
X = validate_data(
self,
X,
ensure_2d=True,
reset=False,
copy=self.copy,
dtype=np.float64,
)

return (X - self.mean_) @ self.components_.T

def predict(self, X: np.ndarray) -> np.ndarray:
"""
Predict new data

Parameters
----------
X : np.ndarray of shape (n_samples, n_features)
The input data to predict.

Returns
-------
y_hat : np.ndarray of shape (n_samples,) or (n_samples, n_targets)
The predicted value.
"""
# Validate input data
check_is_fitted(self, ["mean_", "components_", "coef_", "intercept_"])
X = validate_data(
self,
X,
ensure_2d=True,
reset=False,
copy=self.copy,
dtype=np.float64,
)

# Transform
T = (X - self.mean_) @ self.components_.T
return T @ self.coef_.T + self.intercept_
Loading
Loading