diff --git a/docs/user_guide/selection/ProbeFeatureSelection.rst b/docs/user_guide/selection/ProbeFeatureSelection.rst index 557a462ba..427eb9725 100644 --- a/docs/user_guide/selection/ProbeFeatureSelection.rst +++ b/docs/user_guide/selection/ProbeFeatureSelection.rst @@ -348,6 +348,7 @@ The previous command returns the following output: concave points error 0.003548 fractal dimension error 0.003576 gaussian_probe_0 0.003783 + dtype: float64 Dropping features from the data ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -633,6 +634,102 @@ importance of the probes, as follows: ).fit(X_train, y_train) +With polars +~~~~~~~~~~~ + +:class:`ProbeFeatureSelection()` also works with polars dataframes. The probe features +are returned as a polars dataframe, and the transformed data is a polars dataframe as well. +The feature importance and its standard deviation are returned as dictionaries, with the +feature names as keys. + +Let's load the breast cancer data into a polars dataframe and split it as we did before: + +.. code:: python + + import polars as pl + + data = load_breast_cancer() + X = pl.DataFrame(data.data, schema=list(data.feature_names)) + y = pl.Series("target", data.target) + + X_train, X_test, y_train, y_test = train_test_split( + X, y, test_size=0.2, random_state=3 + ) + +Now, we select features with a random forest and a single probe feature, like we did at +the beginning of this section: + +.. code:: python + + sel = ProbeFeatureSelection( + estimator=RandomForestClassifier(), + scoring="precision", + n_probes=1, + distribution="normal", + cv=5, + random_state=150, + ) + + sel.fit(X_train, y_train) + + sel.probe_features_.head() + +The probe feature has the same values that we obtained with pandas: + +.. code:: text + + shape: (5, 1) + ┌──────────────────┐ + │ gaussian_probe_0 │ + │ --- │ + │ f64 │ + ╞══════════════════╡ + │ -0.69415 │ + │ 1.17184 │ + │ 1.074892 │ + │ 1.698733 │ + │ 0.498702 │ + └──────────────────┘ + +We can read the importance of any feature from the dictionary: + +.. code:: python + + sel.feature_importances_["gaussian_probe_0"] + +which returns the importance of the probe: + +.. code:: python + + 0.003782984070348287 + +And, as with pandas, six features will be removed from the data: + +.. code:: python + + sel.features_to_drop_ + +.. code:: python + + ['mean symmetry', + 'mean fractal dimension', + 'texture error', + 'smoothness error', + 'concave points error', + 'fractal dimension error'] + +Finally, we remove those features from the test set, and obtain a polars dataframe: + +.. code:: python + + Xtr = sel.transform(X_test) + + type(Xtr), Xtr.shape + +.. code:: python + + (, (114, 24)) + Additional resources -------------------- diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/selection/probe_feature_selection.py b/feature_engine/selection/probe_feature_selection.py index f19c5ca76..8e0081c5a 100644 --- a/feature_engine/selection/probe_feature_selection.py +++ b/feature_engine/selection/probe_feature_selection.py @@ -1,7 +1,9 @@ -from typing import List, Union +from typing import Dict, List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -29,6 +31,7 @@ from feature_engine.tags import _return_tags from .base_selection_functions import ( + _importance_series, _select_numerical_variables, find_feature_importance, single_feature_performance, @@ -56,7 +59,7 @@ class ProbeFeatureSelection(BaseSelector): """ ProbeFeatureSelection() generates one or more probe (i.e., random) features based - on a user-selected distribution. The distribution options are 'normal', 'binomial', + on a user-selected distribution. The distribution options are 'normal', 'binary', 'uniform', 'discrete_uniform', 'poisson', or 'all'. 'all' creates `n_probes` of each of the five aforementioned distributions. @@ -98,8 +101,8 @@ class ProbeFeatureSelection(BaseSelector): distribution: str, list, default='normal' The distribution used to create the probe features. The options are 'normal', - 'binomial', 'uniform', 'discrete_uniform', 'poisson' and 'all'. 'all' creates - `n_probes` features per distribution type, i.e., normal, binomial, + 'binary', 'uniform', 'discrete_uniform', 'poisson' and 'all'. 'all' creates + `n_probes` features per distribution type, i.e., normal, binary, uniform, discrete_uniform and poisson. The remaining options create `n_probes` features per selected distributions. @@ -124,17 +127,21 @@ class ProbeFeatureSelection(BaseSelector): ---------- probe_features_: A dataframe comprised of the pseudo-randomly generated features based - on the selected distribution. + on the selected distribution, in the same library as the input (pandas, + polars, ...). feature_importances_: - Pandas Series with the feature importance. If `collective=True`, the feature - importance is given by the coefficients of linear models or the importance - derived from tree-based models. If `collective=False`, the feature importance - is given by a performance metric returned by a model trained using that - individual feature. + The feature importance of the variables and the probe features. A pandas + Series with the features as index when X is a pandas dataframe, and a + dictionary with the features as keys otherwise. If `collective=True`, the + feature importance is given by the coefficients of linear models or the + importance derived from tree-based models. If `collective=False`, the feature + importance is given by a performance metric returned by a model trained using + that individual feature. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. {features_to_drop_} @@ -177,6 +184,27 @@ class ProbeFeatureSelection(BaseSelector): >>> X_tr = sel.fit_transform(X, y) >>> print(X.shape, X_tr.shape) (569, 30) (569, 19) + + With polars: + + >>> import polars as pl + >>> from sklearn.datasets import load_breast_cancer + >>> from sklearn.linear_model import LogisticRegression + >>> from feature_engine.selection import ProbeFeatureSelection + >>> data = load_breast_cancer() + >>> X = pl.DataFrame(data.data, schema=list(data.feature_names)) + >>> y = pl.Series("target", data.target) + >>> sel = ProbeFeatureSelection( + >>> estimator=LogisticRegression(max_iter=1000000), + >>> scoring="roc_auc", + >>> n_probes=3, + >>> distribution="normal", + >>> cv=3, + >>> random_state=150, + >>> ) + >>> X_tr = sel.fit_transform(X, y) + >>> print(X.shape, X_tr.shape) + (569, 30) (569, 19) """ def __init__( @@ -254,19 +282,20 @@ def __init__( self.threshold = threshold self.random_state = random_state - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Find the important features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] + The training dataset. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: array-like of shape (n_samples) Target variable. Required to train the estimator. """ - # check input dataframe - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) self.variables_ = _select_numerical_variables( X, self.variables, self.confirm_variables @@ -275,13 +304,23 @@ def fit(self, X: pd.DataFrame, y: pd.Series): # save input features self._get_feature_names_in(X) - # create probe feature distributions - self.probe_features_ = self._generate_probe_features(X.shape[0]) - - # required for a (train/test) split dataset - X.reset_index(drop=True, inplace=True) + probes = self._generate_probe_features(nw_X.shape[0]) - X_new = pd.concat([X[self.variables_], self.probe_features_], axis=1) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + pd_namespace = nw.get_native_namespace(X) + self.probe_features_ = pd_namespace.DataFrame(probes, copy=False) + # the probes have a default index: X needs the same one to align them. + X_new = pd_namespace.concat( + [X[self.variables_].reset_index(drop=True), self.probe_features_], + axis=1, + ) + else: + nw_probes = nw.from_dict(probes, backend=nw.get_native_namespace(X)) + self.probe_features_ = nw_probes.to_native() + X_new = nw.concat( + [nw_X.select(nw.col(*self.variables_)), nw_probes], how="horizontal" + ).to_native() if self.collective is True: # train model using entire dataset and derive feature importance @@ -298,30 +337,34 @@ def fit(self, X: pd.DataFrame, y: pd.Series): else: # trains a model per feature (single feature models) + features = [*self.variables_, *probes] f_importance_mean, f_importance_std = single_feature_performance( X=X_new, y=y, - variables=X_new.columns, + variables=features, estimator=self.estimator, cv=self.cv, groups=self.groups, scoring=self.scoring, ) - self.feature_importances_ = pd.Series(f_importance_mean) - self.feature_importances_std_ = pd.Series(f_importance_std) + self.feature_importances_ = _importance_series( + X, features, np.array([f_importance_mean[f] for f in features]) + ) + self.feature_importances_std_ = _importance_series( + X, features, np.array([f_importance_std[f] for f in features]) + ) # get features with lower importance than the probe features - self.features_to_drop_ = self._get_features_to_drop() + self.features_to_drop_ = self._get_features_to_drop(list(probes)) return self - def _generate_probe_features(self, n_obs: int) -> pd.DataFrame: + def _generate_probe_features(self, n_obs: int) -> Dict[str, np.ndarray]: """ - Returns a dataframe comprised of the probe features using the - selected distribution. + Returns a dictionary with the name of the probe features as keys and their + values, drawn from the selected distributions, as values. """ - # create dataframe - df = pd.DataFrame() + probes = {} # set random state np.random.seed(self.random_state) @@ -333,58 +376,51 @@ def _generate_probe_features(self, n_obs: int) -> pd.DataFrame: if {"normal", "all"} & distribution: for i in range(self.n_probes): - df[f"gaussian_probe_{i}"] = np.random.normal(0, 3, n_obs) + probes[f"gaussian_probe_{i}"] = np.random.normal(0, 3, n_obs) if {"binary", "all"} & distribution: for i in range(self.n_probes): - df[f"binary_probe_{i}"] = np.random.randint(0, 2, n_obs) + probes[f"binary_probe_{i}"] = np.random.randint(0, 2, n_obs) if {"uniform", "all"} & distribution: for i in range(self.n_probes): - df[f"uniform_probe_{i}"] = np.random.uniform(0, 1, n_obs) + probes[f"uniform_probe_{i}"] = np.random.uniform(0, 1, n_obs) if {"discrete_uniform", "all"} & distribution: for i in range(self.n_probes): - df[f"discrete_uniform_probe_{i}"] = np.random.randint( + probes[f"discrete_uniform_probe_{i}"] = np.random.randint( 0, self.n_categories, n_obs ) if {"poisson", "all"} & distribution: for i in range(self.n_probes): - df[f"poisson_probe_{i}"] = np.random.poisson(self.n_categories, n_obs) + probes[f"poisson_probe_{i}"] = np.random.poisson( + self.n_categories, n_obs + ) - return df + return probes - def _get_features_to_drop(self): + def _get_features_to_drop(self, probes: List[str]) -> List[Union[str, int]]: """ Identify the variables that have a lower feature importance than the average feature importance of all the probe features. """ + probe_importance = np.array( + [self.feature_importances_[probe] for probe in probes] + ) # if more than 1 probe feature, calculate threshold based on # probe feature importance. - if self.probe_features_.shape[1] > 1: + if len(probes) > 1: if self.threshold == "mean": - threshold = self.feature_importances_[ - self.probe_features_.columns - ].values.mean() + threshold = probe_importance.mean() elif self.threshold == "max": - threshold = self.feature_importances_[ - self.probe_features_.columns - ].values.max() + threshold = probe_importance.max() else: - threshold = ( - self.feature_importances_[ - self.probe_features_.columns - ].values.mean() - + 3 - * self.feature_importances_[ - self.probe_features_.columns - ].values.std() - ) + threshold = probe_importance.mean() + 3 * probe_importance.std() else: - threshold = self.feature_importances_[self.probe_features_.columns].values + threshold = probe_importance features_to_drop = [] diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) diff --git a/tests/test_selection/test_probe_feature_selection.py b/tests/test_selection/test_probe_feature_selection.py index 58d489122..651db6558 100644 --- a/tests/test_selection/test_probe_feature_selection.py +++ b/tests/test_selection/test_probe_feature_selection.py @@ -1,598 +1,607 @@ +import re + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier, RandomForestRegressor from sklearn.linear_model import Lasso, LogisticRegression -from sklearn.model_selection import StratifiedKFold, GroupKFold +from sklearn.model_selection import GroupKFold, StratifiedKFold from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor from feature_engine.selection import ProbeFeatureSelection - -_input_params = [ - (RandomForestClassifier(), "precision", "all", 3, 3, 6, 4), - (Lasso(), "neg_mean_squared_error", "binary", 7, 7, 4, 100), - (LogisticRegression(), "roc_auc", "normal", 5, 5, 2, 73), - (DecisionTreeRegressor(), "r2", "uniform", 4, 4, 10, 84), - (DecisionTreeRegressor(), "r2", "discrete_uniform", 4, 4, 10, 84), - (DecisionTreeRegressor(), "r2", "poisson", 4, 4, 10, 84), - (RandomForestClassifier(), "precision", ["binary", "uniform"], 3, 3, 6, 4), -] - - -@pytest.mark.parametrize( - "_estimator, _scoring, _distribution, _n_cat, _cv, _n_probes, _random_state", - _input_params, -) -def test_input_params_assignment( - _estimator, _scoring, _distribution, _n_cat, _cv, _n_probes, _random_state -): - sel = ProbeFeatureSelection( - estimator=_estimator, - scoring=_scoring, - distribution=_distribution, - n_categories=_n_cat, - cv=_cv, - n_probes=_n_probes, - random_state=_random_state, - ) - - assert sel.estimator == _estimator - assert sel.scoring == _scoring - assert sel.distribution == _distribution - assert sel.n_categories == _n_cat - assert sel.cv == _cv - assert sel.n_probes == _n_probes - assert sel.random_state == _random_state - - -@pytest.mark.parametrize("collective", [True, False]) -def test_collective_param(collective): - tr = ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - collective=collective, +from tests.backend_helpers import frame_to_dict, make_series + +VARIABLES = [f"var_{i}" for i in range(12)] + +# importance of the variables and probes of the random forest in test_fit_transform +FOREST_IMPORTANCE = { + "var_0": 0.03, + "var_1": 0, + "var_2": 0, + "var_3": 0, + "var_4": 0.26, + "var_5": 0, + "var_6": 0.22, + "var_7": 0.33, + "var_8": 0.02, + "var_9": 0.12, + "var_10": 0, + "var_11": 0, + "gaussian_probe_0": 0, + "gaussian_probe_1": 0, +} + +# the first rows of the probes drawn with random_state=3 +GAUSSIAN_PROBES = { + "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], + "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _forest_selector(cv=3): + # the estimator is not seeded: fit() seeds numpy's global random generator. + return ProbeFeatureSelection( + estimator=RandomForestClassifier(), + distribution="normal", + n_probes=2, + scoring="recall", + cv=cv, + random_state=3, ) - assert tr.collective is collective -@pytest.mark.parametrize("collective", [10, "string", 0.1]) -def test_collective_raises_error(collective): +# init parameters +@pytest.mark.parametrize("collective", [10, "string", 0.1, None, [True]]) +def test_error_if_collective_not_bool(collective): msg = f"collective takes values True or False. Got {collective} instead." - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - collective=collective, - ) + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), collective=collective) @pytest.mark.parametrize( - "_distribution", [3, "distribution", ["salud", "binary"], 2.22] + "distribution", [3, 2.22, None, "distribution", ["salud", "binary"], ("normal",)] ) -def test_raises_error_when_not_permitted_distribution(_distribution): - with pytest.raises(ValueError): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - distribution=_distribution, - ) - - -@pytest.mark.parametrize("_n_probes", ["tree", [False, 2], 101.1]) -def test_raises_error_when_not_permitted_n_probes(_n_probes): - with pytest.raises(ValueError): +def test_error_if_distribution_not_permitted(distribution): + msg = ( + "distribution takes values 'normal', 'binary', 'uniform', " + f"'discrete_uniform', 'poisson', or 'all'. Got {distribution} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_probes=_n_probes, + estimator=DecisionTreeRegressor(), distribution=distribution ) -@pytest.mark.parametrize("n_cat", [0.1, "string", 0, -1]) -def test_n_categories_raises_error(n_cat): - msg = f"n_categories must be a positive integer. Got {n_cat} instead." - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_categories=n_cat, - ) +@pytest.mark.parametrize("n_probes", ["tree", [False, 2], 101.1, None]) +def test_error_if_n_probes_not_int(n_probes): + msg = f"n_probes must be an integer. Got {n_probes} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), n_probes=n_probes) -@pytest.mark.parametrize("n_cat", [[10], [10, 1], {10}, {10, 1}]) -def test_n_categories_raises_error_with_collections(n_cat): - # I had problems with collections and string matching so - # I split the test - with pytest.raises(ValueError): +@pytest.mark.parametrize( + "n_categories", [0.1, "string", 0, -1, None, [10], [10, 1], {10}] +) +def test_error_if_n_categories_not_positive_int(n_categories): + msg = f"n_categories must be a positive integer. Got {n_categories} instead." + with pytest.raises(ValueError, match=re.escape(msg)): ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_categories=n_cat, + estimator=DecisionTreeRegressor(), n_categories=n_categories ) -@pytest.mark.parametrize("thresh", ["mean", "max", "mean_plus_std"]) -def test_threshold_init_param(thresh): - sel = ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - threshold=thresh, - ) - assert sel.threshold == thresh - - -@pytest.mark.parametrize("thresh", [1, "string"]) -def test_threshold_raises_error(thresh): +@pytest.mark.parametrize("threshold", [1, "string", None, ["mean"]]) +def test_error_if_threshold_not_permitted(threshold): msg = ( "threshold takes values 'mean', 'max' or 'mean_plus_std'. " - f"Got {thresh} instead." + f"Got {threshold} instead." ) - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - threshold=thresh, - ) - + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), threshold=threshold) -def test_fit_transform_functionality(df_test): - X, y = df_test +@pytest.mark.parametrize( + "estimator, collective, scoring, n_probes, distribution, n_categories, " + "threshold, cv, groups, random_state, confirm_variables", + [ + ( + RandomForestClassifier(), + True, + "precision", + 3, + "all", + 3, + "mean", + 3, + None, + 4, + False, + ), + ( + Lasso(), + False, + "neg_mean_squared_error", + 7, + "binary", + 7, + "max", + StratifiedKFold(), + [1, 2], + None, + True, + ), + ( + DecisionTreeRegressor(), + True, + "r2", + 1, + ["binary", "uniform"], + 10, + "mean_plus_std", + GroupKFold(), + None, + 84, + False, + ), + ], +) +def test_init_param_assignment( + estimator, + collective, + scoring, + n_probes, + distribution, + n_categories, + threshold, + cv, + groups, + random_state, + confirm_variables, +): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, + estimator=estimator, + collective=collective, + scoring=scoring, + n_probes=n_probes, + distribution=distribution, + n_categories=n_categories, + threshold=threshold, + cv=cv, + groups=groups, + random_state=random_state, + confirm_variables=confirm_variables, ) - X_tr = sel.fit_transform(X, y) + assert sel.estimator is estimator + assert sel.collective is collective + assert sel.scoring == scoring + assert sel.n_probes == n_probes + assert sel.distribution == distribution + assert sel.n_categories == n_categories + assert sel.threshold == threshold + assert sel.cv is cv + assert sel.groups == groups + assert sel.random_state == random_state + assert sel.confirm_variables is confirm_variables - # expected results - expected_probe_features = { - "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], - "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], - } - expected_probe_features_df = pd.DataFrame(expected_probe_features) - expected_feature_importances = pd.Series( - data=[0.03, 0, 0, 0, 0.26, 0, 0.22, 0.33, 0.02, 0.12, 0, 0, 0, 0], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], +# fit and transform +def test_fit_transform(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + + sel = _forest_selector() + Xt = sel.fit_transform(X, y) + + assert isinstance(sel.probe_features_, make_df) + probes = frame_to_dict(sel.probe_features_) + assert {k: [round(x, 3) for x in v[:5]] for k, v in probes.items()} == ( + GAUSSIAN_PROBES ) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, + assert len(probes["gaussian_probe_0"]) == 1000 + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) + assert sel.variables_ == VARIABLES assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False - ) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + var: data_classification[var] + for var in VARIABLES + if var not in ["var_2", "var_10"] + } -def test_generate_some_probe_features(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, - distribution=["normal", "binary", "uniform"], - random_state=1, - ) +def test_feature_importances_are_series_for_pandas_and_dict_otherwise( + make_df, data_classification +): + X, y = _split_target(make_df, data_classification) + sel = _forest_selector().fit(X, y) - probe_features = sel._generate_probe_features(5).round(3) + container = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(sel.feature_importances_, container) + assert isinstance(sel.feature_importances_std_, container) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], - "gaussian_probe_1": [-6.905, 5.234, -2.284, 0.957, -0.748], - "binary_probe_0": [1, 0, 0, 0, 1], - "binary_probe_1": [1, 1, 1, 1, 0], - "uniform_probe_0": [0.198, 0.801, 0.968, 0.313, 0.692], - "uniform_probe_1": [0.876, 0.895, 0.085, 0.039, 0.170], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False - ) +@pytest.mark.parametrize("cv_type", ["splitter", "splits"]) +def test_cv_as_splitter_or_splits(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = StratifiedKFold(n_splits=3) + if cv_type == "splits": + cv = cv.split(np.zeros(1000), data_classification["target"]) + sel = _forest_selector(cv=cv).fit(X, y) -def test_generate_all_probe_features(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution="all", - random_state=1, + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) + assert sel.features_to_drop_ == ["var_2", "var_10"] - probe_features = sel._generate_probe_features(5).round(3) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], - "binary_probe_0": [0, 1, 0, 0, 1], - "uniform_probe_0": [0.443, 0.230, 0.534, 0.914, 0.457], - "discrete_uniform_probe_0": [1, 7, 0, 6, 9], - "poisson_probe_0": [19, 15, 2, 14, 10], - } - expected_results_df = pd.DataFrame(expected_results) +def test_feature_importance_std(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = _forest_selector().fit(X, y) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + assert dict(sel.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0088, + "var_1": 0.0002, + "var_2": 0.0005, + "var_3": 0.0007, + "var_4": 0.0343, + "var_5": 0.0013, + "var_6": 0.0089, + "var_7": 0.0551, + "var_8": 0.0049, + "var_9": 0.0123, + "var_10": 0.0005, + "var_11": 0.0005, + "gaussian_probe_0": 0.0005, + "gaussian_probe_1": 0.0006, + }, + abs=5e-5, ) -def test_generate_probe_features_normal(): +def test_single_feature_models(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, + estimator=RandomForestClassifier(n_estimators=3, random_state=3), distribution="normal", - random_state=1, - ) + collective=False, + n_probes=2, + scoring="recall", + cv=3, + random_state=3, + ).fit(X, y) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 0.5867, + "var_1": 0.5342, + "var_2": 0.5042, + "var_3": 0.4941, + "var_4": 0.9456, + "var_5": 0.5081, + "var_6": 0.9294, + "var_7": 0.9859, + "var_8": 0.4799, + "var_9": 0.8972, + "var_10": 0.4476, + "var_11": 0.5544, + "gaussian_probe_0": 0.4799, + "gaussian_probe_1": 0.5323, + }, + abs=5e-5, + ) + assert sel.features_to_drop_ == ["var_2", "var_3", "var_8", "var_10"] + + +def test_single_feature_models_with_all_distributions(make_df, data_classification): + X, y = _split_target(make_df, data_classification) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585], - "gaussian_probe_1": [-3.219, 2.596, -6.905], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + sel = ProbeFeatureSelection( + estimator=DecisionTreeClassifier(max_depth=2, random_state=0), + collective=False, + scoring="accuracy", + distribution="all", + threshold="mean_plus_std", + cv=3, + random_state=1, ) + Xt = sel.fit_transform(X, y) + + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 0.589, + "var_1": 0.516, + "var_2": 0.502, + "var_3": 0.475, + "var_4": 0.962, + "var_5": 0.486, + "var_6": 0.961, + "var_7": 0.992, + "var_8": 0.519, + "var_9": 0.945, + "var_10": 0.486, + "var_11": 0.518, + "gaussian_probe_0": 0.49, + "binary_probe_0": 0.513, + "uniform_probe_0": 0.508, + "discrete_uniform_probe_0": 0.463, + "poisson_probe_0": 0.499, + }, + abs=5e-4, + ) + kept = ["var_0", "var_4", "var_6", "var_7", "var_9"] + assert sel.features_to_drop_ == [var for var in VARIABLES if var not in kept] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {var: data_classification[var] for var in kept} + + +def test_examines_only_the_indicated_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) - -def test_generate_probe_features_binary(): sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=3, - distribution="binary", + estimator=LogisticRegression(), + variables=["var_0", "var_2", "var_4", "var_7", "var_10"], + distribution=["binary", "uniform"], + n_probes=2, + threshold="max", + cv=3, random_state=1, ) - - n_obs = 2 - probe_features = sel._generate_probe_features(n_obs) - - # expected results - expected_results = { - "binary_probe_0": [1, 1], - "binary_probe_1": [0, 0], - "binary_probe_2": [1, 1], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + Xt = sel.fit_transform(X, y) + + assert list(sel.probe_features_.columns) == [ + "binary_probe_0", + "binary_probe_1", + "uniform_probe_0", + "uniform_probe_1", + ] + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 2.0089, + "var_2": 0.097, + "var_4": 0.3998, + "var_7": 2.6832, + "var_10": 0.1582, + "binary_probe_0": 0.268, + "binary_probe_1": 0.3385, + "uniform_probe_0": 0.0804, + "uniform_probe_1": 0.2257, + }, + abs=5e-5, ) + assert sel.features_to_drop_ == ["var_2", "var_10"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [v for v in VARIABLES if v not in ["var_2", "var_10"]] -def test_generate_probe_features_uniform(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution="uniform", - random_state=1, - ) +def test_non_numerical_variables_are_not_examined(make_df, data_classification): + data = {**data_classification, "city": ["London", "Paris"] * 500} + X, y = _split_target(make_df, data) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + sel = _forest_selector() + Xt = sel.fit_transform(X, y) - # expected results - expected_results = {"uniform_probe_0": [0.417, 0.72, 0.0]} - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False - ) + assert sel.variables_ == VARIABLES + assert sel.features_to_drop_ == ["var_2", "var_10"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [ + v for v in VARIABLES + ["city"] if v not in ["var_2", "var_10"] + ] -def test_generate_probe_features_discrete(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, - distribution=["discrete_uniform", "poisson"], - random_state=1, - ) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_classification, to_target): + X, _ = _split_target(make_df, data_classification) + y = to_target(data_classification["target"]) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + sel = _forest_selector().fit(X, y) - # expected results - expected_results = { - "discrete_uniform_probe_0": [5, 8, 9], - "discrete_uniform_probe_1": [5, 0, 0], - "poisson_probe_0": [9, 14, 10], - "poisson_probe_1": [7, 15, 13], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) + assert sel.features_to_drop_ == ["var_2", "var_10"] -_params = [ - (5, 4, 12), - (10, 9, 19), - (3, 2, 8), -] +def test_groups_give_the_same_result_as_the_group_splits( + make_df, data_classification +): + X, y = _split_target(make_df, data_classification) + groups = np.repeat(np.arange(10), 100) + params = dict( + estimator=RandomForestRegressor(n_estimators=3, random_state=3), + n_probes=2, + scoring="neg_mean_absolute_error", + random_state=3, + ) + splits = GroupKFold(n_splits=3).split(np.zeros(1000), groups=groups) + sel_splits = ProbeFeatureSelection(cv=splits, **params).fit(X, y) + sel = ProbeFeatureSelection(cv=GroupKFold(n_splits=3), groups=groups, **params) + Xt = sel.fit_transform(X, y) -@pytest.mark.parametrize("n_cats, max_discrete, max_poisson", _params) -def test_generate_features_n_categories(n_cats, max_discrete, max_poisson): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution=["discrete_uniform", "poisson"], - n_categories=n_cats, - random_state=1, - ) + assert dict(sel.feature_importances_) == dict(sel_splits.feature_importances_) + assert sel.features_to_drop_ == sel_splits.features_to_drop_ + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == frame_to_dict(sel_splits.transform(X)) - probe_features = sel._generate_probe_features(100) - assert probe_features["discrete_uniform_probe_0"].max() == max_discrete - assert probe_features["poisson_probe_0"].max() == max_poisson +def test_integer_column_names(): + # pandas only: polars does not allow integer column names. + X = pd.DataFrame({0: [0.0, 0.0, 1.0, 1.0] * 25, 1: [0.0, 1.0] * 50}) + y = pd.Series([0, 1] * 50) -@pytest.mark.parametrize("thresh", ["mean", "max", "mean_plus_std"]) -def test_get_features_to_drop_with_one_probe(thresh): sel = ProbeFeatureSelection( - estimator=LogisticRegression(), n_probes=1, threshold=thresh + estimator=DecisionTreeClassifier(max_depth=2, random_state=0), + collective=False, + cv=2, + random_state=2, ) - sel.feature_importances_ = pd.Series( - [11, 12, 9, 10], index=["var1", "var2", "var3", "probe"] + Xt = sel.fit_transform(X, y) + + assert sel.variables_ == [0, 1] + pd.testing.assert_series_equal( + sel.feature_importances_, + pd.Series([0.5, 1.0, 0.5124], index=[0, 1, "gaussian_probe_0"]), ) - sel.probe_features_ = pd.DataFrame({"probe": [1, 1, 1, 1, 1]}) - sel.variables_ = ["var1", "var2", "var3"] - assert sel._get_features_to_drop() == ["var3"] + assert sel.features_to_drop_ == [0] + pd.testing.assert_frame_equal(Xt, X[[1]]) -_thresholds = [ - ("mean", ["var4"]), - ("max", ["var3", "var4"]), - ("mean_plus_std", ["var1", "var3", "var4"]), -] +def test_fit_does_not_change_the_index_of_the_input(data_classification): + # pandas only: the probes are aligned with X by position, not by index. + X = pd.DataFrame({var: data_classification[var] for var in VARIABLES}) + X.index = np.arange(1000)[::-1] * 3 + 10 + y = pd.Series(data_classification["target"], index=X.index) + X_original = X.copy() + sel = _forest_selector() + Xt = sel.fit_transform(X, y) -@pytest.mark.parametrize("thresh, vars_to_drop", _thresholds) -def test_get_features_to_drop_with_many_probes(thresh, vars_to_drop): - sel = ProbeFeatureSelection( - estimator=LogisticRegression(), - n_probes=2, - threshold=thresh, - ) - sel.feature_importances_ = pd.Series( - [11, 20, 9.9, 8.7, 10, 8], - index=["var1", "var2", "var3", "var4", "probe1", "probe2"], - ) - sel.probe_features_ = pd.DataFrame( - {"probe1": [1, 1, 1, 1, 1], "probe2": [1, 1, 1, 1, 1]} - ) - sel.variables_ = ["var1", "var2", "var3", "var4"] - assert sel._get_features_to_drop() == vars_to_drop + pd.testing.assert_frame_equal(X, X_original) + assert sel.probe_features_.index.equals(pd.RangeIndex(1000)) + assert sel.features_to_drop_ == ["var_2", "var_10"] + pd.testing.assert_frame_equal(Xt, X.drop(columns=["var_2", "var_10"])) -def test_cv_generator(df_test): - X, y = df_test - cv = StratifiedKFold(n_splits=3) +def _round_probes(probes): + return {name: np.round(values, 3).tolist() for name, values in probes.items()} - # expected results - expected_probe_features = { - "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], - "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], - } - expected_probe_features_df = pd.DataFrame(expected_probe_features) - - expected_feature_importances = pd.Series( - data=[0.03, 0, 0, 0, 0.26, 0, 0.22, 0.33, 0.02, 0.12, 0, 0, 0, 0], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], - ) - # splitter passed as such +def test_generate_some_probe_features(): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", + estimator=DecisionTreeClassifier(), n_probes=2, - scoring="recall", - cv=cv, - random_state=3, - confirm_variables=False, + distribution=["normal", "binary", "uniform"], + random_state=1, ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, - ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) - assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False - ) + assert _round_probes(sel._generate_probe_features(5)) == { + "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], + "gaussian_probe_1": [-6.905, 5.234, -2.284, 0.957, -0.748], + "binary_probe_0": [1, 0, 0, 0, 1], + "binary_probe_1": [1, 1, 1, 1, 0], + "uniform_probe_0": [0.198, 0.801, 0.968, 0.313, 0.692], + "uniform_probe_1": [0.876, 0.895, 0.085, 0.039, 0.170], + } - # splitter passed as splits - sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=cv.split(X, y), - random_state=3, - confirm_variables=False, - ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, - ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) - assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False +def test_generate_all_probe_features(): + sel = ProbeFeatureSelection( + estimator=DecisionTreeClassifier(), + n_probes=1, + distribution="all", + random_state=1, ) + assert _round_probes(sel._generate_probe_features(5)) == { + "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], + "binary_probe_0": [0, 1, 0, 0, 1], + "uniform_probe_0": [0.443, 0.230, 0.534, 0.914, 0.457], + "discrete_uniform_probe_0": [1, 7, 0, 6, 9], + "poisson_probe_0": [19, 15, 2, 14, 10], + } -def test_feature_importance_std(df_test): - X, y = df_test +@pytest.mark.parametrize( + "distribution, n_probes, n_obs, expected", + [ + ( + "normal", + 2, + 3, + { + "gaussian_probe_0": [4.873, -1.835, -1.585], + "gaussian_probe_1": [-3.219, 2.596, -6.905], + }, + ), + ( + "binary", + 3, + 2, + { + "binary_probe_0": [1, 1], + "binary_probe_1": [0, 0], + "binary_probe_2": [1, 1], + }, + ), + ("uniform", 1, 3, {"uniform_probe_0": [0.417, 0.72, 0.0]}), + ( + ["discrete_uniform", "poisson"], + 2, + 3, + { + "discrete_uniform_probe_0": [5, 8, 9], + "discrete_uniform_probe_1": [5, 0, 0], + "poisson_probe_0": [9, 14, 10], + "poisson_probe_1": [7, 15, 13], + }, + ), + ], +) +def test_generate_probe_features_per_distribution( + distribution, n_probes, n_obs, expected +): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, - ).fit(X, y) - - # expected results - expected_std = pd.Series( - data=[ - 0.0088, - 0.0002, - 0.0005, - 0.0007, - 0.0343, - 0.0013, - 0.0089, - 0.0551, - 0.0049, - 0.0123, - 0.0005, - 0.0005, - 0.0005, - 0.0006, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], + estimator=DecisionTreeClassifier(), + n_probes=n_probes, + distribution=distribution, + random_state=1, ) - assert sel.feature_importances_std_.round(4).equals(expected_std) - + assert _round_probes(sel._generate_probe_features(n_obs)) == expected -def test_single_feature_importance_generation(df_test): - X, y = df_test +@pytest.mark.parametrize( + "n_categories, max_discrete, max_poisson", [(5, 4, 12), (10, 9, 19), (3, 2, 8)] +) +def test_generate_features_n_categories(n_categories, max_discrete, max_poisson): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(n_estimators=3, random_state=3), - distribution="normal", - collective=False, - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, - ).fit(X, y) - - # expected results - expected_ = pd.Series( - data=[ - 0.5867, - 0.5342, - 0.5042, - 0.4941, - 0.9456, - 0.5081, - 0.9294, - 0.9859, - 0.4799, - 0.8972, - 0.4476, - 0.5544, - 0.4799, - 0.5323, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], + estimator=DecisionTreeClassifier(), + n_probes=1, + distribution=["discrete_uniform", "poisson"], + n_categories=n_categories, + random_state=1, ) - assert sel.feature_importances_.round(4).equals(expected_) - + probes = sel._generate_probe_features(100) + assert probes["discrete_uniform_probe_0"].max() == max_discrete + assert probes["poisson_probe_0"].max() == max_poisson -def test_probe_feature_selector_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) - estimator = RandomForestRegressor(n_estimators=3, random_state=3) - distribution = "normal" - n_probes = 2 - scoring = "neg_mean_absolute_error" - random_state = 3 - confirm_variables = True - - sel_expected = ProbeFeatureSelection( - estimator=estimator, - distribution=distribution, - n_probes=n_probes, - scoring=scoring, - cv=cv_indices, - random_state=random_state, - confirm_variables=confirm_variables, +@pytest.mark.parametrize("container", [dict, pd.Series]) +@pytest.mark.parametrize("threshold", ["mean", "max", "mean_plus_std"]) +def test_get_features_to_drop_with_one_probe(container, threshold): + sel = ProbeFeatureSelection(estimator=LogisticRegression(), threshold=threshold) + sel.feature_importances_ = container( + {"var1": 11, "var2": 12, "var3": 9, "probe": 10} ) - X_tr_expected = sel_expected.fit_transform(X, y) + sel.variables_ = ["var1", "var2", "var3"] + assert sel._get_features_to_drop(["probe"]) == ["var3"] - sel = ProbeFeatureSelection( - estimator=estimator, - distribution=distribution, - n_probes=n_probes, - scoring=scoring, - cv=cv, - groups=groups, - random_state=random_state, - confirm_variables=confirm_variables, - ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal(X_tr_expected, X_tr) +@pytest.mark.parametrize("container", [dict, pd.Series]) +@pytest.mark.parametrize( + "threshold, features_to_drop", + [ + ("mean", ["var4"]), + ("max", ["var3", "var4"]), + ("mean_plus_std", ["var1", "var3", "var4"]), + ], +) +def test_get_features_to_drop_with_many_probes( + container, threshold, features_to_drop +): + sel = ProbeFeatureSelection(estimator=LogisticRegression(), threshold=threshold) + sel.feature_importances_ = container( + {"var1": 11, "var2": 20, "var3": 9.9, "var4": 8.7, "probe1": 10, "probe2": 8} + ) + sel.variables_ = ["var1", "var2", "var3", "var4"] + assert sel._get_features_to_drop(["probe1", "probe2"]) == features_to_drop