From cd7f30b2759817b3f0722bcbf8b51d436e1bfec0 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:27:15 +0200 Subject: [PATCH 1/3] Migrate the selection base classes and helpers to narwhals, add polars support BaseSelector.transform() returns the retained features in the train set order, in the same library as the input (pandas X[features], narwhals select otherwise). BaseRecursiveSelector.fit() trains the estimators on native frames and returns (nw_X, y). The helpers in base_selection_functions no longer import pandas: correlations are computed with numpy (np.corrcoef, or matrix products for pairwise complete observations when there are missing values), and feature importances are pandas Series for pandas input and dicts otherwise. Co-Authored-By: Claude Opus 5 --- .../selection/base_recursive_selector.py | 96 ++-- .../selection/base_selection_functions.py | 212 ++++++-- feature_engine/selection/base_selector.py | 44 +- tests/test_selection/conftest.py | 17 + .../test_base_recursive_selector.py | 257 +++++++++ .../test_base_selection_functions.py | 513 ++++++++++-------- tests/test_selection/test_base_selector.py | 178 ++++-- 7 files changed, 924 insertions(+), 393 deletions(-) create mode 100644 tests/test_selection/test_base_recursive_selector.py diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) From 440cc945c054ce609c9d336a52845c88d440f5d3 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:25:11 +0200 Subject: [PATCH 2/3] Migrate ProbeFeatureSelection to narwhals, add polars support fit() no longer resets the index of the input in place: pandas input gets the probes in a new frame (DataFrame(copy=False) + concat, faster than narwhals), other backends append them with a narwhals horizontal concat. Probe values are identical for a given random_state. probe_features_ is a dataframe in the input library; feature importances are a pandas Series for pandas input and a dict otherwise, as in the base selectors. Tests rewritten to run on pandas and polars; user guide and docstring get a polars example. Co-Authored-By: Claude Opus 5 --- .../selection/ProbeFeatureSelection.rst | 97 ++ .../selection/probe_feature_selection.py | 146 ++- .../test_probe_feature_selection.py | 983 +++++++++--------- 3 files changed, 684 insertions(+), 542 deletions(-) diff --git a/docs/user_guide/selection/ProbeFeatureSelection.rst b/docs/user_guide/selection/ProbeFeatureSelection.rst index 557a462ba..802a44767 100644 --- a/docs/user_guide/selection/ProbeFeatureSelection.rst +++ b/docs/user_guide/selection/ProbeFeatureSelection.rst @@ -348,6 +348,7 @@ The previous command returns the following output: concave points error 0.003548 fractal dimension error 0.003576 gaussian_probe_0 0.003783 + dtype: float64 Dropping features from the data ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -633,6 +634,102 @@ importance of the probes, as follows: ).fit(X_train, y_train) +With polars +~~~~~~~~~~~ + +:class:`ProbeFeatureSelection()` also works with polars dataframes. The probe features +are returned as a polars dataframe, and the transformed data is a polars dataframe as well. +The feature importance and its standard deviation are returned as dictionaries, with the +feature names as keys. + +Let's load the breast cancer data into a polars dataframe and split it as we did before: + +.. code:: python + + import polars as pl + + data = load_breast_cancer() + X = pl.DataFrame(data.data, schema=list(data.feature_names)) + y = pl.Series("target", data.target) + + X_train, X_test, y_train, y_test = train_test_split( + X, y, test_size=0.2, random_state=3 + ) + +Now, we select features with a random forest and a single probe feature, like we did at +the beginning of this section: + +.. code:: python + + sel = ProbeFeatureSelection( + estimator=RandomForestClassifier(), + scoring="precision", + n_probes=1, + distribution="normal", + cv=5, + random_state=150, + ) + + sel.fit(X_train, y_train) + + sel.probe_features_.head() + +The probe feature has the same values that we obtained with pandas: + +.. code:: python + + shape: (5, 1) + ┌──────────────────┐ + │ gaussian_probe_0 │ + │ --- │ + │ f64 │ + ╞══════════════════╡ + │ -0.69415 │ + │ 1.17184 │ + │ 1.074892 │ + │ 1.698733 │ + │ 0.498702 │ + └──────────────────┘ + +We can read the importance of any feature from the dictionary: + +.. code:: python + + sel.feature_importances_["gaussian_probe_0"] + +which returns the importance of the probe: + +.. code:: python + + 0.003782984070348287 + +And, as with pandas, six features will be removed from the data: + +.. code:: python + + sel.features_to_drop_ + +.. code:: python + + ['mean symmetry', + 'mean fractal dimension', + 'texture error', + 'smoothness error', + 'concave points error', + 'fractal dimension error'] + +Finally, we remove those features from the test set, and obtain a polars dataframe: + +.. code:: python + + Xtr = sel.transform(X_test) + + type(Xtr), Xtr.shape + +.. code:: python + + (, (114, 24)) + Additional resources -------------------- diff --git a/feature_engine/selection/probe_feature_selection.py b/feature_engine/selection/probe_feature_selection.py index f19c5ca76..8e0081c5a 100644 --- a/feature_engine/selection/probe_feature_selection.py +++ b/feature_engine/selection/probe_feature_selection.py @@ -1,7 +1,9 @@ -from typing import List, Union +from typing import Dict, List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -29,6 +31,7 @@ from feature_engine.tags import _return_tags from .base_selection_functions import ( + _importance_series, _select_numerical_variables, find_feature_importance, single_feature_performance, @@ -56,7 +59,7 @@ class ProbeFeatureSelection(BaseSelector): """ ProbeFeatureSelection() generates one or more probe (i.e., random) features based - on a user-selected distribution. The distribution options are 'normal', 'binomial', + on a user-selected distribution. The distribution options are 'normal', 'binary', 'uniform', 'discrete_uniform', 'poisson', or 'all'. 'all' creates `n_probes` of each of the five aforementioned distributions. @@ -98,8 +101,8 @@ class ProbeFeatureSelection(BaseSelector): distribution: str, list, default='normal' The distribution used to create the probe features. The options are 'normal', - 'binomial', 'uniform', 'discrete_uniform', 'poisson' and 'all'. 'all' creates - `n_probes` features per distribution type, i.e., normal, binomial, + 'binary', 'uniform', 'discrete_uniform', 'poisson' and 'all'. 'all' creates + `n_probes` features per distribution type, i.e., normal, binary, uniform, discrete_uniform and poisson. The remaining options create `n_probes` features per selected distributions. @@ -124,17 +127,21 @@ class ProbeFeatureSelection(BaseSelector): ---------- probe_features_: A dataframe comprised of the pseudo-randomly generated features based - on the selected distribution. + on the selected distribution, in the same library as the input (pandas, + polars, ...). feature_importances_: - Pandas Series with the feature importance. If `collective=True`, the feature - importance is given by the coefficients of linear models or the importance - derived from tree-based models. If `collective=False`, the feature importance - is given by a performance metric returned by a model trained using that - individual feature. + The feature importance of the variables and the probe features. A pandas + Series with the features as index when X is a pandas dataframe, and a + dictionary with the features as keys otherwise. If `collective=True`, the + feature importance is given by the coefficients of linear models or the + importance derived from tree-based models. If `collective=False`, the feature + importance is given by a performance metric returned by a model trained using + that individual feature. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. {features_to_drop_} @@ -177,6 +184,27 @@ class ProbeFeatureSelection(BaseSelector): >>> X_tr = sel.fit_transform(X, y) >>> print(X.shape, X_tr.shape) (569, 30) (569, 19) + + With polars: + + >>> import polars as pl + >>> from sklearn.datasets import load_breast_cancer + >>> from sklearn.linear_model import LogisticRegression + >>> from feature_engine.selection import ProbeFeatureSelection + >>> data = load_breast_cancer() + >>> X = pl.DataFrame(data.data, schema=list(data.feature_names)) + >>> y = pl.Series("target", data.target) + >>> sel = ProbeFeatureSelection( + >>> estimator=LogisticRegression(max_iter=1000000), + >>> scoring="roc_auc", + >>> n_probes=3, + >>> distribution="normal", + >>> cv=3, + >>> random_state=150, + >>> ) + >>> X_tr = sel.fit_transform(X, y) + >>> print(X.shape, X_tr.shape) + (569, 30) (569, 19) """ def __init__( @@ -254,19 +282,20 @@ def __init__( self.threshold = threshold self.random_state = random_state - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Find the important features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] + The training dataset. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: array-like of shape (n_samples) Target variable. Required to train the estimator. """ - # check input dataframe - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) self.variables_ = _select_numerical_variables( X, self.variables, self.confirm_variables @@ -275,13 +304,23 @@ def fit(self, X: pd.DataFrame, y: pd.Series): # save input features self._get_feature_names_in(X) - # create probe feature distributions - self.probe_features_ = self._generate_probe_features(X.shape[0]) - - # required for a (train/test) split dataset - X.reset_index(drop=True, inplace=True) + probes = self._generate_probe_features(nw_X.shape[0]) - X_new = pd.concat([X[self.variables_], self.probe_features_], axis=1) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + pd_namespace = nw.get_native_namespace(X) + self.probe_features_ = pd_namespace.DataFrame(probes, copy=False) + # the probes have a default index: X needs the same one to align them. + X_new = pd_namespace.concat( + [X[self.variables_].reset_index(drop=True), self.probe_features_], + axis=1, + ) + else: + nw_probes = nw.from_dict(probes, backend=nw.get_native_namespace(X)) + self.probe_features_ = nw_probes.to_native() + X_new = nw.concat( + [nw_X.select(nw.col(*self.variables_)), nw_probes], how="horizontal" + ).to_native() if self.collective is True: # train model using entire dataset and derive feature importance @@ -298,30 +337,34 @@ def fit(self, X: pd.DataFrame, y: pd.Series): else: # trains a model per feature (single feature models) + features = [*self.variables_, *probes] f_importance_mean, f_importance_std = single_feature_performance( X=X_new, y=y, - variables=X_new.columns, + variables=features, estimator=self.estimator, cv=self.cv, groups=self.groups, scoring=self.scoring, ) - self.feature_importances_ = pd.Series(f_importance_mean) - self.feature_importances_std_ = pd.Series(f_importance_std) + self.feature_importances_ = _importance_series( + X, features, np.array([f_importance_mean[f] for f in features]) + ) + self.feature_importances_std_ = _importance_series( + X, features, np.array([f_importance_std[f] for f in features]) + ) # get features with lower importance than the probe features - self.features_to_drop_ = self._get_features_to_drop() + self.features_to_drop_ = self._get_features_to_drop(list(probes)) return self - def _generate_probe_features(self, n_obs: int) -> pd.DataFrame: + def _generate_probe_features(self, n_obs: int) -> Dict[str, np.ndarray]: """ - Returns a dataframe comprised of the probe features using the - selected distribution. + Returns a dictionary with the name of the probe features as keys and their + values, drawn from the selected distributions, as values. """ - # create dataframe - df = pd.DataFrame() + probes = {} # set random state np.random.seed(self.random_state) @@ -333,58 +376,51 @@ def _generate_probe_features(self, n_obs: int) -> pd.DataFrame: if {"normal", "all"} & distribution: for i in range(self.n_probes): - df[f"gaussian_probe_{i}"] = np.random.normal(0, 3, n_obs) + probes[f"gaussian_probe_{i}"] = np.random.normal(0, 3, n_obs) if {"binary", "all"} & distribution: for i in range(self.n_probes): - df[f"binary_probe_{i}"] = np.random.randint(0, 2, n_obs) + probes[f"binary_probe_{i}"] = np.random.randint(0, 2, n_obs) if {"uniform", "all"} & distribution: for i in range(self.n_probes): - df[f"uniform_probe_{i}"] = np.random.uniform(0, 1, n_obs) + probes[f"uniform_probe_{i}"] = np.random.uniform(0, 1, n_obs) if {"discrete_uniform", "all"} & distribution: for i in range(self.n_probes): - df[f"discrete_uniform_probe_{i}"] = np.random.randint( + probes[f"discrete_uniform_probe_{i}"] = np.random.randint( 0, self.n_categories, n_obs ) if {"poisson", "all"} & distribution: for i in range(self.n_probes): - df[f"poisson_probe_{i}"] = np.random.poisson(self.n_categories, n_obs) + probes[f"poisson_probe_{i}"] = np.random.poisson( + self.n_categories, n_obs + ) - return df + return probes - def _get_features_to_drop(self): + def _get_features_to_drop(self, probes: List[str]) -> List[Union[str, int]]: """ Identify the variables that have a lower feature importance than the average feature importance of all the probe features. """ + probe_importance = np.array( + [self.feature_importances_[probe] for probe in probes] + ) # if more than 1 probe feature, calculate threshold based on # probe feature importance. - if self.probe_features_.shape[1] > 1: + if len(probes) > 1: if self.threshold == "mean": - threshold = self.feature_importances_[ - self.probe_features_.columns - ].values.mean() + threshold = probe_importance.mean() elif self.threshold == "max": - threshold = self.feature_importances_[ - self.probe_features_.columns - ].values.max() + threshold = probe_importance.max() else: - threshold = ( - self.feature_importances_[ - self.probe_features_.columns - ].values.mean() - + 3 - * self.feature_importances_[ - self.probe_features_.columns - ].values.std() - ) + threshold = probe_importance.mean() + 3 * probe_importance.std() else: - threshold = self.feature_importances_[self.probe_features_.columns].values + threshold = probe_importance features_to_drop = [] diff --git a/tests/test_selection/test_probe_feature_selection.py b/tests/test_selection/test_probe_feature_selection.py index 58d489122..651db6558 100644 --- a/tests/test_selection/test_probe_feature_selection.py +++ b/tests/test_selection/test_probe_feature_selection.py @@ -1,598 +1,607 @@ +import re + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier, RandomForestRegressor from sklearn.linear_model import Lasso, LogisticRegression -from sklearn.model_selection import StratifiedKFold, GroupKFold +from sklearn.model_selection import GroupKFold, StratifiedKFold from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor from feature_engine.selection import ProbeFeatureSelection - -_input_params = [ - (RandomForestClassifier(), "precision", "all", 3, 3, 6, 4), - (Lasso(), "neg_mean_squared_error", "binary", 7, 7, 4, 100), - (LogisticRegression(), "roc_auc", "normal", 5, 5, 2, 73), - (DecisionTreeRegressor(), "r2", "uniform", 4, 4, 10, 84), - (DecisionTreeRegressor(), "r2", "discrete_uniform", 4, 4, 10, 84), - (DecisionTreeRegressor(), "r2", "poisson", 4, 4, 10, 84), - (RandomForestClassifier(), "precision", ["binary", "uniform"], 3, 3, 6, 4), -] - - -@pytest.mark.parametrize( - "_estimator, _scoring, _distribution, _n_cat, _cv, _n_probes, _random_state", - _input_params, -) -def test_input_params_assignment( - _estimator, _scoring, _distribution, _n_cat, _cv, _n_probes, _random_state -): - sel = ProbeFeatureSelection( - estimator=_estimator, - scoring=_scoring, - distribution=_distribution, - n_categories=_n_cat, - cv=_cv, - n_probes=_n_probes, - random_state=_random_state, - ) - - assert sel.estimator == _estimator - assert sel.scoring == _scoring - assert sel.distribution == _distribution - assert sel.n_categories == _n_cat - assert sel.cv == _cv - assert sel.n_probes == _n_probes - assert sel.random_state == _random_state - - -@pytest.mark.parametrize("collective", [True, False]) -def test_collective_param(collective): - tr = ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - collective=collective, +from tests.backend_helpers import frame_to_dict, make_series + +VARIABLES = [f"var_{i}" for i in range(12)] + +# importance of the variables and probes of the random forest in test_fit_transform +FOREST_IMPORTANCE = { + "var_0": 0.03, + "var_1": 0, + "var_2": 0, + "var_3": 0, + "var_4": 0.26, + "var_5": 0, + "var_6": 0.22, + "var_7": 0.33, + "var_8": 0.02, + "var_9": 0.12, + "var_10": 0, + "var_11": 0, + "gaussian_probe_0": 0, + "gaussian_probe_1": 0, +} + +# the first rows of the probes drawn with random_state=3 +GAUSSIAN_PROBES = { + "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], + "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _forest_selector(cv=3): + # the estimator is not seeded: fit() seeds numpy's global random generator. + return ProbeFeatureSelection( + estimator=RandomForestClassifier(), + distribution="normal", + n_probes=2, + scoring="recall", + cv=cv, + random_state=3, ) - assert tr.collective is collective -@pytest.mark.parametrize("collective", [10, "string", 0.1]) -def test_collective_raises_error(collective): +# init parameters +@pytest.mark.parametrize("collective", [10, "string", 0.1, None, [True]]) +def test_error_if_collective_not_bool(collective): msg = f"collective takes values True or False. Got {collective} instead." - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - collective=collective, - ) + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), collective=collective) @pytest.mark.parametrize( - "_distribution", [3, "distribution", ["salud", "binary"], 2.22] + "distribution", [3, 2.22, None, "distribution", ["salud", "binary"], ("normal",)] ) -def test_raises_error_when_not_permitted_distribution(_distribution): - with pytest.raises(ValueError): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - distribution=_distribution, - ) - - -@pytest.mark.parametrize("_n_probes", ["tree", [False, 2], 101.1]) -def test_raises_error_when_not_permitted_n_probes(_n_probes): - with pytest.raises(ValueError): +def test_error_if_distribution_not_permitted(distribution): + msg = ( + "distribution takes values 'normal', 'binary', 'uniform', " + f"'discrete_uniform', 'poisson', or 'all'. Got {distribution} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_probes=_n_probes, + estimator=DecisionTreeRegressor(), distribution=distribution ) -@pytest.mark.parametrize("n_cat", [0.1, "string", 0, -1]) -def test_n_categories_raises_error(n_cat): - msg = f"n_categories must be a positive integer. Got {n_cat} instead." - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_categories=n_cat, - ) +@pytest.mark.parametrize("n_probes", ["tree", [False, 2], 101.1, None]) +def test_error_if_n_probes_not_int(n_probes): + msg = f"n_probes must be an integer. Got {n_probes} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), n_probes=n_probes) -@pytest.mark.parametrize("n_cat", [[10], [10, 1], {10}, {10, 1}]) -def test_n_categories_raises_error_with_collections(n_cat): - # I had problems with collections and string matching so - # I split the test - with pytest.raises(ValueError): +@pytest.mark.parametrize( + "n_categories", [0.1, "string", 0, -1, None, [10], [10, 1], {10}] +) +def test_error_if_n_categories_not_positive_int(n_categories): + msg = f"n_categories must be a positive integer. Got {n_categories} instead." + with pytest.raises(ValueError, match=re.escape(msg)): ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - n_categories=n_cat, + estimator=DecisionTreeRegressor(), n_categories=n_categories ) -@pytest.mark.parametrize("thresh", ["mean", "max", "mean_plus_std"]) -def test_threshold_init_param(thresh): - sel = ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - threshold=thresh, - ) - assert sel.threshold == thresh - - -@pytest.mark.parametrize("thresh", [1, "string"]) -def test_threshold_raises_error(thresh): +@pytest.mark.parametrize("threshold", [1, "string", None, ["mean"]]) +def test_error_if_threshold_not_permitted(threshold): msg = ( "threshold takes values 'mean', 'max' or 'mean_plus_std'. " - f"Got {thresh} instead." + f"Got {threshold} instead." ) - with pytest.raises(ValueError, match=msg): - ProbeFeatureSelection( - estimator=DecisionTreeRegressor(), - threshold=thresh, - ) - + with pytest.raises(ValueError, match=re.escape(msg)): + ProbeFeatureSelection(estimator=DecisionTreeRegressor(), threshold=threshold) -def test_fit_transform_functionality(df_test): - X, y = df_test +@pytest.mark.parametrize( + "estimator, collective, scoring, n_probes, distribution, n_categories, " + "threshold, cv, groups, random_state, confirm_variables", + [ + ( + RandomForestClassifier(), + True, + "precision", + 3, + "all", + 3, + "mean", + 3, + None, + 4, + False, + ), + ( + Lasso(), + False, + "neg_mean_squared_error", + 7, + "binary", + 7, + "max", + StratifiedKFold(), + [1, 2], + None, + True, + ), + ( + DecisionTreeRegressor(), + True, + "r2", + 1, + ["binary", "uniform"], + 10, + "mean_plus_std", + GroupKFold(), + None, + 84, + False, + ), + ], +) +def test_init_param_assignment( + estimator, + collective, + scoring, + n_probes, + distribution, + n_categories, + threshold, + cv, + groups, + random_state, + confirm_variables, +): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, + estimator=estimator, + collective=collective, + scoring=scoring, + n_probes=n_probes, + distribution=distribution, + n_categories=n_categories, + threshold=threshold, + cv=cv, + groups=groups, + random_state=random_state, + confirm_variables=confirm_variables, ) - X_tr = sel.fit_transform(X, y) + assert sel.estimator is estimator + assert sel.collective is collective + assert sel.scoring == scoring + assert sel.n_probes == n_probes + assert sel.distribution == distribution + assert sel.n_categories == n_categories + assert sel.threshold == threshold + assert sel.cv is cv + assert sel.groups == groups + assert sel.random_state == random_state + assert sel.confirm_variables is confirm_variables - # expected results - expected_probe_features = { - "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], - "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], - } - expected_probe_features_df = pd.DataFrame(expected_probe_features) - expected_feature_importances = pd.Series( - data=[0.03, 0, 0, 0, 0.26, 0, 0.22, 0.33, 0.02, 0.12, 0, 0, 0, 0], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], +# fit and transform +def test_fit_transform(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + + sel = _forest_selector() + Xt = sel.fit_transform(X, y) + + assert isinstance(sel.probe_features_, make_df) + probes = frame_to_dict(sel.probe_features_) + assert {k: [round(x, 3) for x in v[:5]] for k, v in probes.items()} == ( + GAUSSIAN_PROBES ) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, + assert len(probes["gaussian_probe_0"]) == 1000 + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) + assert sel.variables_ == VARIABLES assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False - ) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + var: data_classification[var] + for var in VARIABLES + if var not in ["var_2", "var_10"] + } -def test_generate_some_probe_features(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, - distribution=["normal", "binary", "uniform"], - random_state=1, - ) +def test_feature_importances_are_series_for_pandas_and_dict_otherwise( + make_df, data_classification +): + X, y = _split_target(make_df, data_classification) + sel = _forest_selector().fit(X, y) - probe_features = sel._generate_probe_features(5).round(3) + container = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(sel.feature_importances_, container) + assert isinstance(sel.feature_importances_std_, container) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], - "gaussian_probe_1": [-6.905, 5.234, -2.284, 0.957, -0.748], - "binary_probe_0": [1, 0, 0, 0, 1], - "binary_probe_1": [1, 1, 1, 1, 0], - "uniform_probe_0": [0.198, 0.801, 0.968, 0.313, 0.692], - "uniform_probe_1": [0.876, 0.895, 0.085, 0.039, 0.170], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False - ) +@pytest.mark.parametrize("cv_type", ["splitter", "splits"]) +def test_cv_as_splitter_or_splits(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = StratifiedKFold(n_splits=3) + if cv_type == "splits": + cv = cv.split(np.zeros(1000), data_classification["target"]) + sel = _forest_selector(cv=cv).fit(X, y) -def test_generate_all_probe_features(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution="all", - random_state=1, + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) + assert sel.features_to_drop_ == ["var_2", "var_10"] - probe_features = sel._generate_probe_features(5).round(3) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], - "binary_probe_0": [0, 1, 0, 0, 1], - "uniform_probe_0": [0.443, 0.230, 0.534, 0.914, 0.457], - "discrete_uniform_probe_0": [1, 7, 0, 6, 9], - "poisson_probe_0": [19, 15, 2, 14, 10], - } - expected_results_df = pd.DataFrame(expected_results) +def test_feature_importance_std(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = _forest_selector().fit(X, y) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + assert dict(sel.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0088, + "var_1": 0.0002, + "var_2": 0.0005, + "var_3": 0.0007, + "var_4": 0.0343, + "var_5": 0.0013, + "var_6": 0.0089, + "var_7": 0.0551, + "var_8": 0.0049, + "var_9": 0.0123, + "var_10": 0.0005, + "var_11": 0.0005, + "gaussian_probe_0": 0.0005, + "gaussian_probe_1": 0.0006, + }, + abs=5e-5, ) -def test_generate_probe_features_normal(): +def test_single_feature_models(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, + estimator=RandomForestClassifier(n_estimators=3, random_state=3), distribution="normal", - random_state=1, - ) + collective=False, + n_probes=2, + scoring="recall", + cv=3, + random_state=3, + ).fit(X, y) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 0.5867, + "var_1": 0.5342, + "var_2": 0.5042, + "var_3": 0.4941, + "var_4": 0.9456, + "var_5": 0.5081, + "var_6": 0.9294, + "var_7": 0.9859, + "var_8": 0.4799, + "var_9": 0.8972, + "var_10": 0.4476, + "var_11": 0.5544, + "gaussian_probe_0": 0.4799, + "gaussian_probe_1": 0.5323, + }, + abs=5e-5, + ) + assert sel.features_to_drop_ == ["var_2", "var_3", "var_8", "var_10"] + + +def test_single_feature_models_with_all_distributions(make_df, data_classification): + X, y = _split_target(make_df, data_classification) - # expected results - expected_results = { - "gaussian_probe_0": [4.873, -1.835, -1.585], - "gaussian_probe_1": [-3.219, 2.596, -6.905], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + sel = ProbeFeatureSelection( + estimator=DecisionTreeClassifier(max_depth=2, random_state=0), + collective=False, + scoring="accuracy", + distribution="all", + threshold="mean_plus_std", + cv=3, + random_state=1, ) + Xt = sel.fit_transform(X, y) + + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 0.589, + "var_1": 0.516, + "var_2": 0.502, + "var_3": 0.475, + "var_4": 0.962, + "var_5": 0.486, + "var_6": 0.961, + "var_7": 0.992, + "var_8": 0.519, + "var_9": 0.945, + "var_10": 0.486, + "var_11": 0.518, + "gaussian_probe_0": 0.49, + "binary_probe_0": 0.513, + "uniform_probe_0": 0.508, + "discrete_uniform_probe_0": 0.463, + "poisson_probe_0": 0.499, + }, + abs=5e-4, + ) + kept = ["var_0", "var_4", "var_6", "var_7", "var_9"] + assert sel.features_to_drop_ == [var for var in VARIABLES if var not in kept] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {var: data_classification[var] for var in kept} + + +def test_examines_only_the_indicated_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) - -def test_generate_probe_features_binary(): sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=3, - distribution="binary", + estimator=LogisticRegression(), + variables=["var_0", "var_2", "var_4", "var_7", "var_10"], + distribution=["binary", "uniform"], + n_probes=2, + threshold="max", + cv=3, random_state=1, ) - - n_obs = 2 - probe_features = sel._generate_probe_features(n_obs) - - # expected results - expected_results = { - "binary_probe_0": [1, 1], - "binary_probe_1": [0, 0], - "binary_probe_2": [1, 1], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + Xt = sel.fit_transform(X, y) + + assert list(sel.probe_features_.columns) == [ + "binary_probe_0", + "binary_probe_1", + "uniform_probe_0", + "uniform_probe_1", + ] + assert dict(sel.feature_importances_) == pytest.approx( + { + "var_0": 2.0089, + "var_2": 0.097, + "var_4": 0.3998, + "var_7": 2.6832, + "var_10": 0.1582, + "binary_probe_0": 0.268, + "binary_probe_1": 0.3385, + "uniform_probe_0": 0.0804, + "uniform_probe_1": 0.2257, + }, + abs=5e-5, ) + assert sel.features_to_drop_ == ["var_2", "var_10"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [v for v in VARIABLES if v not in ["var_2", "var_10"]] -def test_generate_probe_features_uniform(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution="uniform", - random_state=1, - ) +def test_non_numerical_variables_are_not_examined(make_df, data_classification): + data = {**data_classification, "city": ["London", "Paris"] * 500} + X, y = _split_target(make_df, data) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + sel = _forest_selector() + Xt = sel.fit_transform(X, y) - # expected results - expected_results = {"uniform_probe_0": [0.417, 0.72, 0.0]} - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False - ) + assert sel.variables_ == VARIABLES + assert sel.features_to_drop_ == ["var_2", "var_10"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [ + v for v in VARIABLES + ["city"] if v not in ["var_2", "var_10"] + ] -def test_generate_probe_features_discrete(): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=2, - distribution=["discrete_uniform", "poisson"], - random_state=1, - ) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_classification, to_target): + X, _ = _split_target(make_df, data_classification) + y = to_target(data_classification["target"]) - n_obs = 3 - probe_features = sel._generate_probe_features(n_obs).round(3) + sel = _forest_selector().fit(X, y) - # expected results - expected_results = { - "discrete_uniform_probe_0": [5, 8, 9], - "discrete_uniform_probe_1": [5, 0, 0], - "poisson_probe_0": [9, 14, 10], - "poisson_probe_1": [7, 15, 13], - } - expected_results_df = pd.DataFrame(expected_results) - pd.testing.assert_frame_equal( - probe_features, expected_results_df, check_dtype=False + assert dict(sel.feature_importances_) == pytest.approx( + FOREST_IMPORTANCE, abs=5e-3 ) + assert sel.features_to_drop_ == ["var_2", "var_10"] -_params = [ - (5, 4, 12), - (10, 9, 19), - (3, 2, 8), -] +def test_groups_give_the_same_result_as_the_group_splits( + make_df, data_classification +): + X, y = _split_target(make_df, data_classification) + groups = np.repeat(np.arange(10), 100) + params = dict( + estimator=RandomForestRegressor(n_estimators=3, random_state=3), + n_probes=2, + scoring="neg_mean_absolute_error", + random_state=3, + ) + splits = GroupKFold(n_splits=3).split(np.zeros(1000), groups=groups) + sel_splits = ProbeFeatureSelection(cv=splits, **params).fit(X, y) + sel = ProbeFeatureSelection(cv=GroupKFold(n_splits=3), groups=groups, **params) + Xt = sel.fit_transform(X, y) -@pytest.mark.parametrize("n_cats, max_discrete, max_poisson", _params) -def test_generate_features_n_categories(n_cats, max_discrete, max_poisson): - sel = ProbeFeatureSelection( - estimator=DecisionTreeClassifier(), - n_probes=1, - distribution=["discrete_uniform", "poisson"], - n_categories=n_cats, - random_state=1, - ) + assert dict(sel.feature_importances_) == dict(sel_splits.feature_importances_) + assert sel.features_to_drop_ == sel_splits.features_to_drop_ + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == frame_to_dict(sel_splits.transform(X)) - probe_features = sel._generate_probe_features(100) - assert probe_features["discrete_uniform_probe_0"].max() == max_discrete - assert probe_features["poisson_probe_0"].max() == max_poisson +def test_integer_column_names(): + # pandas only: polars does not allow integer column names. + X = pd.DataFrame({0: [0.0, 0.0, 1.0, 1.0] * 25, 1: [0.0, 1.0] * 50}) + y = pd.Series([0, 1] * 50) -@pytest.mark.parametrize("thresh", ["mean", "max", "mean_plus_std"]) -def test_get_features_to_drop_with_one_probe(thresh): sel = ProbeFeatureSelection( - estimator=LogisticRegression(), n_probes=1, threshold=thresh + estimator=DecisionTreeClassifier(max_depth=2, random_state=0), + collective=False, + cv=2, + random_state=2, ) - sel.feature_importances_ = pd.Series( - [11, 12, 9, 10], index=["var1", "var2", "var3", "probe"] + Xt = sel.fit_transform(X, y) + + assert sel.variables_ == [0, 1] + pd.testing.assert_series_equal( + sel.feature_importances_, + pd.Series([0.5, 1.0, 0.5124], index=[0, 1, "gaussian_probe_0"]), ) - sel.probe_features_ = pd.DataFrame({"probe": [1, 1, 1, 1, 1]}) - sel.variables_ = ["var1", "var2", "var3"] - assert sel._get_features_to_drop() == ["var3"] + assert sel.features_to_drop_ == [0] + pd.testing.assert_frame_equal(Xt, X[[1]]) -_thresholds = [ - ("mean", ["var4"]), - ("max", ["var3", "var4"]), - ("mean_plus_std", ["var1", "var3", "var4"]), -] +def test_fit_does_not_change_the_index_of_the_input(data_classification): + # pandas only: the probes are aligned with X by position, not by index. + X = pd.DataFrame({var: data_classification[var] for var in VARIABLES}) + X.index = np.arange(1000)[::-1] * 3 + 10 + y = pd.Series(data_classification["target"], index=X.index) + X_original = X.copy() + sel = _forest_selector() + Xt = sel.fit_transform(X, y) -@pytest.mark.parametrize("thresh, vars_to_drop", _thresholds) -def test_get_features_to_drop_with_many_probes(thresh, vars_to_drop): - sel = ProbeFeatureSelection( - estimator=LogisticRegression(), - n_probes=2, - threshold=thresh, - ) - sel.feature_importances_ = pd.Series( - [11, 20, 9.9, 8.7, 10, 8], - index=["var1", "var2", "var3", "var4", "probe1", "probe2"], - ) - sel.probe_features_ = pd.DataFrame( - {"probe1": [1, 1, 1, 1, 1], "probe2": [1, 1, 1, 1, 1]} - ) - sel.variables_ = ["var1", "var2", "var3", "var4"] - assert sel._get_features_to_drop() == vars_to_drop + pd.testing.assert_frame_equal(X, X_original) + assert sel.probe_features_.index.equals(pd.RangeIndex(1000)) + assert sel.features_to_drop_ == ["var_2", "var_10"] + pd.testing.assert_frame_equal(Xt, X.drop(columns=["var_2", "var_10"])) -def test_cv_generator(df_test): - X, y = df_test - cv = StratifiedKFold(n_splits=3) +def _round_probes(probes): + return {name: np.round(values, 3).tolist() for name, values in probes.items()} - # expected results - expected_probe_features = { - "gaussian_probe_0": [5.366, 1.31, 0.289, -5.59, -0.832], - "gaussian_probe_1": [0.104, 3.396, -7.67, -0.807, -5.729], - } - expected_probe_features_df = pd.DataFrame(expected_probe_features) - - expected_feature_importances = pd.Series( - data=[0.03, 0, 0, 0, 0.26, 0, 0.22, 0.33, 0.02, 0.12, 0, 0, 0, 0], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], - ) - # splitter passed as such +def test_generate_some_probe_features(): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", + estimator=DecisionTreeClassifier(), n_probes=2, - scoring="recall", - cv=cv, - random_state=3, - confirm_variables=False, + distribution=["normal", "binary", "uniform"], + random_state=1, ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, - ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) - assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False - ) + assert _round_probes(sel._generate_probe_features(5)) == { + "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], + "gaussian_probe_1": [-6.905, 5.234, -2.284, 0.957, -0.748], + "binary_probe_0": [1, 0, 0, 0, 1], + "binary_probe_1": [1, 1, 1, 1, 0], + "uniform_probe_0": [0.198, 0.801, 0.968, 0.313, 0.692], + "uniform_probe_1": [0.876, 0.895, 0.085, 0.039, 0.170], + } - # splitter passed as splits - sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=cv.split(X, y), - random_state=3, - confirm_variables=False, - ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal( - sel.probe_features_.head().round(3), - expected_probe_features_df, - check_dtype=False, - ) - assert sel.feature_importances_.round(2).equals(expected_feature_importances) - assert sel.features_to_drop_ == ["var_2", "var_10"] - pd.testing.assert_frame_equal( - X_tr, X.drop(columns=["var_2", "var_10"]), check_dtype=False +def test_generate_all_probe_features(): + sel = ProbeFeatureSelection( + estimator=DecisionTreeClassifier(), + n_probes=1, + distribution="all", + random_state=1, ) + assert _round_probes(sel._generate_probe_features(5)) == { + "gaussian_probe_0": [4.873, -1.835, -1.585, -3.219, 2.596], + "binary_probe_0": [0, 1, 0, 0, 1], + "uniform_probe_0": [0.443, 0.230, 0.534, 0.914, 0.457], + "discrete_uniform_probe_0": [1, 7, 0, 6, 9], + "poisson_probe_0": [19, 15, 2, 14, 10], + } -def test_feature_importance_std(df_test): - X, y = df_test +@pytest.mark.parametrize( + "distribution, n_probes, n_obs, expected", + [ + ( + "normal", + 2, + 3, + { + "gaussian_probe_0": [4.873, -1.835, -1.585], + "gaussian_probe_1": [-3.219, 2.596, -6.905], + }, + ), + ( + "binary", + 3, + 2, + { + "binary_probe_0": [1, 1], + "binary_probe_1": [0, 0], + "binary_probe_2": [1, 1], + }, + ), + ("uniform", 1, 3, {"uniform_probe_0": [0.417, 0.72, 0.0]}), + ( + ["discrete_uniform", "poisson"], + 2, + 3, + { + "discrete_uniform_probe_0": [5, 8, 9], + "discrete_uniform_probe_1": [5, 0, 0], + "poisson_probe_0": [9, 14, 10], + "poisson_probe_1": [7, 15, 13], + }, + ), + ], +) +def test_generate_probe_features_per_distribution( + distribution, n_probes, n_obs, expected +): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(), - distribution="normal", - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, - ).fit(X, y) - - # expected results - expected_std = pd.Series( - data=[ - 0.0088, - 0.0002, - 0.0005, - 0.0007, - 0.0343, - 0.0013, - 0.0089, - 0.0551, - 0.0049, - 0.0123, - 0.0005, - 0.0005, - 0.0005, - 0.0006, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], + estimator=DecisionTreeClassifier(), + n_probes=n_probes, + distribution=distribution, + random_state=1, ) - assert sel.feature_importances_std_.round(4).equals(expected_std) - + assert _round_probes(sel._generate_probe_features(n_obs)) == expected -def test_single_feature_importance_generation(df_test): - X, y = df_test +@pytest.mark.parametrize( + "n_categories, max_discrete, max_poisson", [(5, 4, 12), (10, 9, 19), (3, 2, 8)] +) +def test_generate_features_n_categories(n_categories, max_discrete, max_poisson): sel = ProbeFeatureSelection( - estimator=RandomForestClassifier(n_estimators=3, random_state=3), - distribution="normal", - collective=False, - n_probes=2, - scoring="recall", - cv=3, - random_state=3, - confirm_variables=False, - ).fit(X, y) - - # expected results - expected_ = pd.Series( - data=[ - 0.5867, - 0.5342, - 0.5042, - 0.4941, - 0.9456, - 0.5081, - 0.9294, - 0.9859, - 0.4799, - 0.8972, - 0.4476, - 0.5544, - 0.4799, - 0.5323, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - "gaussian_probe_0", - "gaussian_probe_1", - ], + estimator=DecisionTreeClassifier(), + n_probes=1, + distribution=["discrete_uniform", "poisson"], + n_categories=n_categories, + random_state=1, ) - assert sel.feature_importances_.round(4).equals(expected_) - + probes = sel._generate_probe_features(100) + assert probes["discrete_uniform_probe_0"].max() == max_discrete + assert probes["poisson_probe_0"].max() == max_poisson -def test_probe_feature_selector_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) - estimator = RandomForestRegressor(n_estimators=3, random_state=3) - distribution = "normal" - n_probes = 2 - scoring = "neg_mean_absolute_error" - random_state = 3 - confirm_variables = True - - sel_expected = ProbeFeatureSelection( - estimator=estimator, - distribution=distribution, - n_probes=n_probes, - scoring=scoring, - cv=cv_indices, - random_state=random_state, - confirm_variables=confirm_variables, +@pytest.mark.parametrize("container", [dict, pd.Series]) +@pytest.mark.parametrize("threshold", ["mean", "max", "mean_plus_std"]) +def test_get_features_to_drop_with_one_probe(container, threshold): + sel = ProbeFeatureSelection(estimator=LogisticRegression(), threshold=threshold) + sel.feature_importances_ = container( + {"var1": 11, "var2": 12, "var3": 9, "probe": 10} ) - X_tr_expected = sel_expected.fit_transform(X, y) + sel.variables_ = ["var1", "var2", "var3"] + assert sel._get_features_to_drop(["probe"]) == ["var3"] - sel = ProbeFeatureSelection( - estimator=estimator, - distribution=distribution, - n_probes=n_probes, - scoring=scoring, - cv=cv, - groups=groups, - random_state=random_state, - confirm_variables=confirm_variables, - ) - X_tr = sel.fit_transform(X, y) - pd.testing.assert_frame_equal(X_tr_expected, X_tr) +@pytest.mark.parametrize("container", [dict, pd.Series]) +@pytest.mark.parametrize( + "threshold, features_to_drop", + [ + ("mean", ["var4"]), + ("max", ["var3", "var4"]), + ("mean_plus_std", ["var1", "var3", "var4"]), + ], +) +def test_get_features_to_drop_with_many_probes( + container, threshold, features_to_drop +): + sel = ProbeFeatureSelection(estimator=LogisticRegression(), threshold=threshold) + sel.feature_importances_ = container( + {"var1": 11, "var2": 20, "var3": 9.9, "var4": 8.7, "probe1": 10, "probe2": 8} + ) + sel.variables_ = ["var1", "var2", "var3", "var4"] + assert sel._get_features_to_drop(["probe1", "probe2"]) == features_to_drop From ac9db720ffba619f4123487043193b94ff076c15 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:50:07 +0200 Subject: [PATCH 3/3] Mark polars output blocks in the user guide as text Co-Authored-By: Claude Opus 5 --- docs/user_guide/selection/ProbeFeatureSelection.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/user_guide/selection/ProbeFeatureSelection.rst b/docs/user_guide/selection/ProbeFeatureSelection.rst index 802a44767..427eb9725 100644 --- a/docs/user_guide/selection/ProbeFeatureSelection.rst +++ b/docs/user_guide/selection/ProbeFeatureSelection.rst @@ -676,7 +676,7 @@ the beginning of this section: The probe feature has the same values that we obtained with pandas: -.. code:: python +.. code:: text shape: (5, 1) ┌──────────────────┐