-
Notifications
You must be signed in to change notification settings - Fork 19
233 feature implementation of pcr #254
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
b9a672b
bdb5c1d
a165aed
1eadd72
59635a1
7cd1f86
edc0fa4
606e60c
fd4f76f
d46d4e4
fd4c6b5
5abb5eb
c353993
12cfcc1
2a57db7
c48398e
ce32cc2
f9526e2
485b586
d0bb9bd
16e04cf
7b9bcc8
daae041
7ee12cb
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change | ||||||||||||||||||||||
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| @@ -0,0 +1,202 @@ | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
| The :mod:`chemotools.models._principal_component_regression` module implements a PCR model. | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| # Authors: Ruggero Guerrini | ||||||||||||||||||||||||
| # License: MIT | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| from sklearn.base import BaseEstimator, TransformerMixin, RegressorMixin | ||||||||||||||||||||||||
| from sklearn.decomposition import PCA | ||||||||||||||||||||||||
| from sklearn.linear_model import LinearRegression | ||||||||||||||||||||||||
| from sklearn.utils.validation import validate_data, check_is_fitted | ||||||||||||||||||||||||
| from numbers import Integral | ||||||||||||||||||||||||
| from sklearn.utils._param_validation import Interval, RealNotInt, StrOptions | ||||||||||||||||||||||||
| import numpy as np | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| class PrincipalComponentRegression(TransformerMixin, RegressorMixin, BaseEstimator): | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
| Implement Principal Component regression | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Parameters | ||||||||||||||||||||||||
| ---------- | ||||||||||||||||||||||||
| n_components : int, default = 2 | ||||||||||||||||||||||||
| The number of components used to calculate the PCA model | ||||||||||||||||||||||||
| # add comments on parameter constraints | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Attributes | ||||||||||||||||||||||||
| ---------- | ||||||||||||||||||||||||
| pca_ : PCA objects from sklearn | ||||||||||||||||||||||||
| Fitted PCA model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| lr_ : Linear Regression objects from sklearn | ||||||||||||||||||||||||
| Fitted Linear Regression model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| components_ : ndarray of shape (n_components, n_features) | ||||||||||||||||||||||||
| Loadings of the PCA model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| explained_variance_ : ndarray of shape (n_components, ) | ||||||||||||||||||||||||
| Explained variance of the PCA model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| explained_variance_ratio_ : ndarray of shape (n_components, ) | ||||||||||||||||||||||||
| Explained variance ratio of the PCA model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| noise_variance_ : float | ||||||||||||||||||||||||
| Equal to the average of (min(n_features, n_samples) - n_components) | ||||||||||||||||||||||||
| smallest eigenvalues of the covariance matrix of X. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| coef_ : ndarray of shape (n_components, ) or (n_targets, n_components) | ||||||||||||||||||||||||
| Coefficients of the Linear Regression model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| intercept_ : ndarray of shape (n_targets, ) | ||||||||||||||||||||||||
| Intercetps of the Linear Regression | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| rank_ : int | ||||||||||||||||||||||||
| Rank from the Linear Regressio model | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| singular_ : ndarray of shape (min(X, y),) | ||||||||||||||||||||||||
| Singular values of X. Only available when X is dense. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Examples | ||||||||||||||||||||||||
| -------- | ||||||||||||||||||||||||
| >>> from chemotools.decomposition import PrincipalComponentRegression | ||||||||||||||||||||||||
| >>> # Generate sample data | ||||||||||||||||||||||||
| >>> X = np.random.randn(100, 50) | ||||||||||||||||||||||||
| >>> X_test = np.random.randn(10, 50) | ||||||||||||||||||||||||
| >>> y = X[:, 0] + 2*X[:, 1] + np.random.randn(100)*0.1 | ||||||||||||||||||||||||
| >>> # Fit model | ||||||||||||||||||||||||
| >>> pcr = PrincipalComponentRegression(n_components=2) | ||||||||||||||||||||||||
| >>> pcr.fit(X, y) | ||||||||||||||||||||||||
| >>> y_hat = pcr.predict(X_test) | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| _parameter_constraints: dict = { | ||||||||||||||||||||||||
| "n_components": [ | ||||||||||||||||||||||||
| Interval(Integral, 0, None, closed="left"), | ||||||||||||||||||||||||
| Interval(RealNotInt, 0, 1, closed="neither"), | ||||||||||||||||||||||||
| StrOptions({"mle"}), | ||||||||||||||||||||||||
| None, | ||||||||||||||||||||||||
| ], | ||||||||||||||||||||||||
| "copy": ["boolean"], | ||||||||||||||||||||||||
| } | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| def __init__( | ||||||||||||||||||||||||
| self, | ||||||||||||||||||||||||
| n_components: int | None = None, | ||||||||||||||||||||||||
| copy: bool = True, | ||||||||||||||||||||||||
| ): | ||||||||||||||||||||||||
| self.n_components = n_components | ||||||||||||||||||||||||
| self.copy = copy | ||||||||||||||||||||||||
|
Comment on lines
+73
to
+89
|
||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| def __sklearn_tags__(self): | ||||||||||||||||||||||||
| tags = super().__sklearn_tags__() | ||||||||||||||||||||||||
| tags.target_tags.multi_output = True | ||||||||||||||||||||||||
| return tags | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| def fit(self, X: np.ndarray, y: np.ndarray) -> "PrincipalComponentRegression": | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
| Fit the model to the input data. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Parameters | ||||||||||||||||||||||||
| ---------- | ||||||||||||||||||||||||
| X : np.ndarray of shape (n_samples, n_features) | ||||||||||||||||||||||||
| The input data to fit the transformer to. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| y : np.ndarray of shape (n_samples, ) or (n_samples, n_targets) | ||||||||||||||||||||||||
| The properties to be predicted. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Returns | ||||||||||||||||||||||||
| ------- | ||||||||||||||||||||||||
| self : PrincipalComponentRegression | ||||||||||||||||||||||||
| The regression model. | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
| # Validate input data | ||||||||||||||||||||||||
| self._validate_params() | ||||||||||||||||||||||||
| X, y = validate_data( | ||||||||||||||||||||||||
| self, | ||||||||||||||||||||||||
| X, | ||||||||||||||||||||||||
| y, | ||||||||||||||||||||||||
| ensure_2d=True, | ||||||||||||||||||||||||
| reset=True, | ||||||||||||||||||||||||
| copy=self.copy, | ||||||||||||||||||||||||
| dtype=np.float64, | ||||||||||||||||||||||||
| multi_output=True, | ||||||||||||||||||||||||
| ) | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| # Train PCA model | ||||||||||||||||||||||||
| pca = PCA(n_components=self.n_components).fit(X) | ||||||||||||||||||||||||
| x_scores = pca.transform(X) | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| # Train linear regression model | ||||||||||||||||||||||||
| lr = LinearRegression().fit(x_scores, y) | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
|
Comment on lines
+126
to
+132
|
||||||||||||||||||||||||
| # Expose fitting attributes for PCA | ||||||||||||||||||||||||
| self.mean_ = pca.mean_ | ||||||||||||||||||||||||
| self.components_ = pca.components_ | ||||||||||||||||||||||||
| self.explained_variance_ = pca.explained_variance_ | ||||||||||||||||||||||||
| self.explained_variance_ratio_ = pca.explained_variance_ratio_ | ||||||||||||||||||||||||
| self.noise_variance_ = pca.noise_variance_ | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| # Expose fitting attributes for linear regression | ||||||||||||||||||||||||
| self.coef_ = lr.coef_ | ||||||||||||||||||||||||
| self.intercept_ = lr.intercept_ | ||||||||||||||||||||||||
| self.rank_ = lr.rank_ | ||||||||||||||||||||||||
| self.singular_ = lr.singular_ | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| return self | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| def transform(self, X: np.ndarray) -> np.ndarray: | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
| Transform new data into the PCA trained latent space | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Parameters | ||||||||||||||||||||||||
| ---------- | ||||||||||||||||||||||||
| X : np.ndarray of shape (n_samples, n_features) | ||||||||||||||||||||||||
| The input data to transform. | ||||||||||||||||||||||||
|
|
||||||||||||||||||||||||
| Returns | ||||||||||||||||||||||||
| ------- | ||||||||||||||||||||||||
| x_scores : np.ndarray of shape (n_samples,n_components) | ||||||||||||||||||||||||
| The transformed data. | ||||||||||||||||||||||||
| """ | ||||||||||||||||||||||||
|
||||||||||||||||||||||||
| """ | |
| """ | |
| check_is_fitted(self, ["pca_", "lr_"]) | |
| X = validate_data( | |
| self, | |
| X, | |
| ensure_2d=True, | |
| reset=False, | |
| copy=self.copy, | |
| dtype=np.float64, | |
| ) |
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
The class docstring has several issues that will surface in generated docs: (1) it states
n_componentsdefault is 2 but__init__defaults toNone; (2) the example import useschemotools.decompositionwhich does not exist in this repo (should bechemotools.models); and (3) it contains a leftover placeholder line# add comments on parameter constraints. Please align the docstring with the actual public API and remove the placeholder.