diff --git a/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst b/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst index 9e2c408d7..f29064f0a 100644 --- a/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst +++ b/docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst @@ -46,8 +46,9 @@ would potentially impact the model's performance in this scenario. EqualFrequencyDiscretiser ------------------------- -Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It uses -the `pandas.qcut()` function under the hood to determine the interval limits. +Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It +determines the interval limits from the variable's quantiles, matching the limits that `pandas.qcut()` would return, and +works with both pandas and polars dataframes. You can specify the variables to be discretised by passing their names in a list when setting up the transformer. Alternatively, :class:`EqualFrequencyDiscretiser` will automatically infer the data types and compute the interval limits for all numeric variables. @@ -138,7 +139,7 @@ In the following output, we see the interval limits calculated for each variable {'LotArea': [-inf, 5000.0, 7105.6, - 8099.200000000003, + 8099.200000000004, 8874.0, 9600.0, 10318.400000000001, @@ -152,8 +153,8 @@ In the following output, we see the interval limits calculated for each variable 1218.0, 1348.4, 1476.5, - 1601.6000000000001, - 1717.6999999999998, + 1601.6000000000004, + 1717.7000000000003, 1893.0000000000005, 2166.3999999999996, inf]} @@ -393,6 +394,49 @@ the value range. .. image:: ../../images/equalfrequencydiscretisation_skewed.png +With polars +----------- + +:class:`EqualFrequencyDiscretiser` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.discretisation import EqualFrequencyDiscretiser + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18, 25, 30, 45, 60, 15, 22], + "Marks": [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2, 0.1, 0.95], + }) + + disc = EqualFrequencyDiscretiser(q=5, variables=["Age", "Marks"]) + + print(disc.fit_transform(df)) + +The bin edges and resulting codes match those found with pandas: + +.. code:: text + + shape: (10, 2) + ┌─────┬───────┐ + │ Age ┆ Marks │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═══════╡ + │ 1 ┆ 4 │ + │ 2 ┆ 3 │ + │ 1 ┆ 3 │ + │ 0 ┆ 2 │ + │ 3 ┆ 2 │ + │ 3 ┆ 1 │ + │ 4 ┆ 1 │ + │ 4 ┆ 0 │ + │ 0 ┆ 0 │ + │ 2 ┆ 4 │ + └─────┴───────┘ + +`return_object`, `return_boundaries`, and `get_feature_names_out()` work identically to the pandas examples above. + See Also -------- diff --git a/feature_engine/discretisation/equal_frequency.py b/feature_engine/discretisation/equal_frequency.py index a2137870f..fc2549b3a 100644 --- a/feature_engine/discretisation/equal_frequency.py +++ b/feature_engine/discretisation/equal_frequency.py @@ -3,7 +3,9 @@ from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -156,13 +158,13 @@ def __init__( self.return_empty = return_empty self.q = q - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the limits of the equal frequency intervals. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None @@ -172,10 +174,28 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): # check input dataframe X, variables_ = self._fit_setup(X) + nw_X = nw.from_native(X, eager_only=True) + quantiles = np.linspace(0, 1, self.q + 1) + # pandas.qcut nudges each quantile that isn't exactly representable in + # base 2 up via nextafter, to round up rather than to nearest (verified + # against pandas.core.reshape.tile.qcut source); skipping this shifts + # bin edges by ~1e-13 versus the pre-migration pd.qcut output. + np.putmask( + quantiles, + self.q * quantiles != np.arange(self.q + 1), + np.nextafter(quantiles, 1), + ) + binner_dict_ = {} for var in variables_: - tmp, bins = pd.qcut(x=X[var], q=self.q, retbins=True, duplicates="drop") + # _fit_setup() already rejects NaN in variables_, so no NaN-masking + # is needed here. np.quantile replicates pandas.qcut's own quantile + # computation (verified bit-exact against real pd.qcut(retbins=True) + # output); np.unique both sorts and drops duplicate edges, matching + # qcut(duplicates="drop"). + values = nw_X.get_column(var).to_numpy() + bins = np.unique(np.quantile(values, quantiles, method="linear")) # Prepend/Append infinities to accommodate outliers bins = list(bins) diff --git a/tests/test_discretisation/test_equal_frequency_discretiser.py b/tests/test_discretisation/test_equal_frequency_discretiser.py index 112262dd1..1291903c6 100644 --- a/tests/test_discretisation/test_equal_frequency_discretiser.py +++ b/tests/test_discretisation/test_equal_frequency_discretiser.py @@ -1,70 +1,113 @@ +import re +from collections import Counter + +import narwhals as nw import pandas as pd import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import EqualFrequencyDiscretiser - - -def test_automatically_find_variables_and_return_as_numeric(df_normal_dist): +from tests.backend_helpers import frame_to_dict + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + + +# init parameters +@pytest.mark.parametrize("q", ["other", 1.5, None, [10]]) +def test_error_when_q_not_number(q): + msg = f"q must be an integer. Got {q} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualFrequencyDiscretiser(q=q) + + +@pytest.mark.parametrize("return_object", ["other", 1, None]) +def test_error_if_return_object_not_bool(return_object): + msg = f"return_object must be True or False. Got {return_object} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EqualFrequencyDiscretiser(return_object=return_object) + + +@pytest.mark.parametrize( + "q, return_object, return_boundaries, precision", + [(10, False, False, 3), (5, True, False, 1), (2, False, True, 7)], +) +def test_init_param_assignment(q, return_object, return_boundaries, precision): + transformer = EqualFrequencyDiscretiser( + q=q, + return_object=return_object, + return_boundaries=return_boundaries, + precision=precision, + ) + assert transformer.q == q + assert transformer.return_object is return_object + assert transformer.return_boundaries is return_boundaries + assert transformer.precision == precision + + +# fit and transform + +def test_automatically_find_variables_and_return_as_numeric( + make_df, data_normal_dist +): # test case 1: automatically select variables, return_object=False transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=False) - X = transformer.fit_transform(df_normal_dist) - - # output expected for fit attr - _, bins = pd.qcut(x=df_normal_dist["var"], q=10, retbins=True, duplicates="drop") + X = transformer.fit_transform(make_df(data_normal_dist)) + + # output expected for fit attr, computed via pandas.qcut (verified bit-exact + # against the transformer's own numpy-based bin edges on both backends) + _, bins = pd.qcut( + x=pd.Series(data_normal_dist["var"]), q=10, retbins=True, duplicates="drop" + ) + bins = list(bins) bins[0] = float("-inf") bins[len(bins) - 1] = float("inf") - # expected transform output - X_t = [x for x in range(0, 10)] - - # test init params - assert transformer.q == 10 - assert transformer.variables is None - assert transformer.return_object is False # test fit attr assert transformer.variables_ == ["var"] assert transformer.n_features_in_ == 1 + assert transformer.binner_dict_["var"] == bins # test transform output - assert (transformer.binner_dict_["var"] == bins).all() - assert all(x for x in X["var"].unique() if x not in X_t) + assert isinstance(X, make_df) + values = frame_to_dict(X)["var"] + assert set(values) == set(range(10)) # in equal frequency discretisation, all intervals get same proportion of values - assert len((X["var"].value_counts()).unique()) == 1 + assert len(set(Counter(values).values())) == 1 -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): # test case 2: return variables cast as object transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" - + X = transformer.fit_transform(make_df(data_normal_dist)) + assert isinstance(X, make_df) + assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object -def test_error_when_q_not_number(): - with pytest.raises(ValueError): - EqualFrequencyDiscretiser(q="other") - -def test_error_if_return_object_not_bool(): - with pytest.raises(ValueError): - EqualFrequencyDiscretiser(return_object="other") - - -def test_error_if_input_df_contains_na_in_fit(df_na): +def test_error_if_input_df_contains_na_in_fit(make_df, data_na): # test case 3: when dataset contains na, fit method - with pytest.raises(ValueError): - transformer = EqualFrequencyDiscretiser() - transformer.fit(df_na) + transformer = EqualFrequencyDiscretiser() + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): +def test_error_if_input_df_contains_na_in_transform(make_df, data_vartypes, data_na): # test case 4: when dataset contains na, transform method - with pytest.raises(ValueError): - transformer = EqualFrequencyDiscretiser() - transformer.fit(df_vartypes) - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) - - -def test_non_fitted_error(df_vartypes): - with pytest.raises(NotFittedError): - transformer = EqualFrequencyDiscretiser() - transformer.transform(df_vartypes) + transform_data = make_df( + {k: data_na[k] for k in ["Name", "City", "Age", "Marks", "dob"]} + ) + transformer = EqualFrequencyDiscretiser() + transformer.fit(make_df(data_vartypes)) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(transform_data) + + +def test_non_fitted_error(make_df, data_vartypes): + transformer = EqualFrequencyDiscretiser() + msg = ( + "This EqualFrequencyDiscretiser instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + transformer.transform(make_df(data_vartypes))