Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 49 additions & 5 deletions docs/user_guide/discretisation/EqualFrequencyDiscretiser.rst
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,9 @@ would potentially impact the model's performance in this scenario.
EqualFrequencyDiscretiser
-------------------------

Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It uses
the `pandas.qcut()` function under the hood to determine the interval limits.
Feature-engine's :class:`EqualFrequencyDiscretiser` applies equal frequency discretisation to numerical variables. It
determines the interval limits from the variable's quantiles, matching the limits that `pandas.qcut()` would return, and
works with both pandas and polars dataframes.

You can specify the variables to be discretised by passing their names in a list when setting up the transformer. Alternatively,
:class:`EqualFrequencyDiscretiser` will automatically infer the data types and compute the interval limits for all numeric variables.
Expand Down Expand Up @@ -138,7 +139,7 @@ In the following output, we see the interval limits calculated for each variable
{'LotArea': [-inf,
5000.0,
7105.6,
8099.200000000003,
8099.200000000004,
8874.0,
9600.0,
10318.400000000001,
Expand All @@ -152,8 +153,8 @@ In the following output, we see the interval limits calculated for each variable
1218.0,
1348.4,
1476.5,
1601.6000000000001,
1717.6999999999998,
1601.6000000000004,
1717.7000000000003,
1893.0000000000005,
2166.3999999999996,
inf]}
Expand Down Expand Up @@ -393,6 +394,49 @@ the value range.

.. image:: ../../images/equalfrequencydiscretisation_skewed.png

With polars
-----------

:class:`EqualFrequencyDiscretiser` works in the same way with a polars dataframe:

.. code:: python

import polars as pl
from feature_engine.discretisation import EqualFrequencyDiscretiser

df = pl.DataFrame({
"Age": [20, 21, 19, 18, 25, 30, 45, 60, 15, 22],
"Marks": [0.9, 0.8, 0.7, 0.6, 0.5, 0.4, 0.3, 0.2, 0.1, 0.95],
})

disc = EqualFrequencyDiscretiser(q=5, variables=["Age", "Marks"])

print(disc.fit_transform(df))

The bin edges and resulting codes match those found with pandas:

.. code:: text

shape: (10, 2)
┌─────┬───────┐
│ Age ┆ Marks │
│ --- ┆ --- │
│ i64 ┆ i64 │
╞═════╪═══════╡
│ 1 ┆ 4 │
│ 2 ┆ 3 │
│ 1 ┆ 3 │
│ 0 ┆ 2 │
│ 3 ┆ 2 │
│ 3 ┆ 1 │
│ 4 ┆ 1 │
│ 4 ┆ 0 │
│ 0 ┆ 0 │
│ 2 ┆ 4 │
└─────┴───────┘

`return_object`, `return_boundaries`, and `get_feature_names_out()` work identically to the pandas examples above.

See Also
--------

Expand Down
28 changes: 24 additions & 4 deletions feature_engine/discretisation/equal_frequency.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,9 @@

from typing import List, Optional, Union

import pandas as pd
import narwhals as nw
import numpy as np
from narwhals.typing import IntoDataFrame, IntoSeries

from feature_engine._check_init_parameters.check_init_input_params import (
_check_return_empty_is_bool,
Expand Down Expand Up @@ -156,13 +158,13 @@ def __init__(
self.return_empty = return_empty
self.q = q

def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None):
def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None):
"""
Learn the limits of the equal frequency intervals.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features]
X: dataframe of shape = [n_samples, n_features]
The training dataset. Can be the entire dataframe, not just the variables
to be transformed.
y: None
Expand All @@ -172,10 +174,28 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None):
# check input dataframe
X, variables_ = self._fit_setup(X)

nw_X = nw.from_native(X, eager_only=True)
quantiles = np.linspace(0, 1, self.q + 1)
# pandas.qcut nudges each quantile that isn't exactly representable in
# base 2 up via nextafter, to round up rather than to nearest (verified
# against pandas.core.reshape.tile.qcut source); skipping this shifts
# bin edges by ~1e-13 versus the pre-migration pd.qcut output.
np.putmask(
quantiles,
self.q * quantiles != np.arange(self.q + 1),
np.nextafter(quantiles, 1),
)

binner_dict_ = {}

for var in variables_:
tmp, bins = pd.qcut(x=X[var], q=self.q, retbins=True, duplicates="drop")
# _fit_setup() already rejects NaN in variables_, so no NaN-masking
# is needed here. np.quantile replicates pandas.qcut's own quantile
# computation (verified bit-exact against real pd.qcut(retbins=True)
# output); np.unique both sorts and drops duplicate edges, matching
# qcut(duplicates="drop").
values = nw_X.get_column(var).to_numpy()
bins = np.unique(np.quantile(values, quantiles, method="linear"))

# Prepend/Append infinities to accommodate outliers
bins = list(bins)
Expand Down
133 changes: 88 additions & 45 deletions tests/test_discretisation/test_equal_frequency_discretiser.py
Original file line number Diff line number Diff line change
@@ -1,70 +1,113 @@
import re
from collections import Counter

import narwhals as nw
import pandas as pd
import pytest
from sklearn.exceptions import NotFittedError

from feature_engine.discretisation import EqualFrequencyDiscretiser


def test_automatically_find_variables_and_return_as_numeric(df_normal_dist):
from tests.backend_helpers import frame_to_dict

MSG_NA = (
"Some of the variables in the dataset contain NaN. Check and "
"remove those before using this transformer."
)


# init parameters
@pytest.mark.parametrize("q", ["other", 1.5, None, [10]])
def test_error_when_q_not_number(q):
msg = f"q must be an integer. Got {q} instead."
with pytest.raises(ValueError, match=re.escape(msg)):
EqualFrequencyDiscretiser(q=q)


@pytest.mark.parametrize("return_object", ["other", 1, None])
def test_error_if_return_object_not_bool(return_object):
msg = f"return_object must be True or False. Got {return_object} instead."
with pytest.raises(ValueError, match=re.escape(msg)):
EqualFrequencyDiscretiser(return_object=return_object)


@pytest.mark.parametrize(
"q, return_object, return_boundaries, precision",
[(10, False, False, 3), (5, True, False, 1), (2, False, True, 7)],
)
def test_init_param_assignment(q, return_object, return_boundaries, precision):
transformer = EqualFrequencyDiscretiser(
q=q,
return_object=return_object,
return_boundaries=return_boundaries,
precision=precision,
)
assert transformer.q == q
assert transformer.return_object is return_object
assert transformer.return_boundaries is return_boundaries
assert transformer.precision == precision


# fit and transform

def test_automatically_find_variables_and_return_as_numeric(
make_df, data_normal_dist
):
# test case 1: automatically select variables, return_object=False
transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=False)
X = transformer.fit_transform(df_normal_dist)

# output expected for fit attr
_, bins = pd.qcut(x=df_normal_dist["var"], q=10, retbins=True, duplicates="drop")
X = transformer.fit_transform(make_df(data_normal_dist))

# output expected for fit attr, computed via pandas.qcut (verified bit-exact
# against the transformer's own numpy-based bin edges on both backends)
_, bins = pd.qcut(
x=pd.Series(data_normal_dist["var"]), q=10, retbins=True, duplicates="drop"
)
bins = list(bins)
bins[0] = float("-inf")
bins[len(bins) - 1] = float("inf")

# expected transform output
X_t = [x for x in range(0, 10)]

# test init params
assert transformer.q == 10
assert transformer.variables is None
assert transformer.return_object is False
# test fit attr
assert transformer.variables_ == ["var"]
assert transformer.n_features_in_ == 1
assert transformer.binner_dict_["var"] == bins
# test transform output
assert (transformer.binner_dict_["var"] == bins).all()
assert all(x for x in X["var"].unique() if x not in X_t)
assert isinstance(X, make_df)
values = frame_to_dict(X)["var"]
assert set(values) == set(range(10))
# in equal frequency discretisation, all intervals get same proportion of values
assert len((X["var"].value_counts()).unique()) == 1
assert len(set(Counter(values).values())) == 1


def test_automatically_find_variables_and_return_as_object(df_normal_dist):
def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist):
# test case 2: return variables cast as object
transformer = EqualFrequencyDiscretiser(q=10, variables=None, return_object=True)
X = transformer.fit_transform(df_normal_dist)
assert X["var"].dtypes == "O"

X = transformer.fit_transform(make_df(data_normal_dist))
assert isinstance(X, make_df)
assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object

def test_error_when_q_not_number():
with pytest.raises(ValueError):
EqualFrequencyDiscretiser(q="other")


def test_error_if_return_object_not_bool():
with pytest.raises(ValueError):
EqualFrequencyDiscretiser(return_object="other")


def test_error_if_input_df_contains_na_in_fit(df_na):
def test_error_if_input_df_contains_na_in_fit(make_df, data_na):
# test case 3: when dataset contains na, fit method
with pytest.raises(ValueError):
transformer = EqualFrequencyDiscretiser()
transformer.fit(df_na)
transformer = EqualFrequencyDiscretiser()
with pytest.raises(ValueError, match=re.escape(MSG_NA)):
transformer.fit(make_df(data_na))


def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na):
def test_error_if_input_df_contains_na_in_transform(make_df, data_vartypes, data_na):
# test case 4: when dataset contains na, transform method
with pytest.raises(ValueError):
transformer = EqualFrequencyDiscretiser()
transformer.fit(df_vartypes)
transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]])


def test_non_fitted_error(df_vartypes):
with pytest.raises(NotFittedError):
transformer = EqualFrequencyDiscretiser()
transformer.transform(df_vartypes)
transform_data = make_df(
{k: data_na[k] for k in ["Name", "City", "Age", "Marks", "dob"]}
)
transformer = EqualFrequencyDiscretiser()
transformer.fit(make_df(data_vartypes))
with pytest.raises(ValueError, match=re.escape(MSG_NA)):
transformer.transform(transform_data)


def test_non_fitted_error(make_df, data_vartypes):
transformer = EqualFrequencyDiscretiser()
msg = (
"This EqualFrequencyDiscretiser instance is not fitted yet. Call 'fit' with "
"appropriate arguments before using this estimator."
)
with pytest.raises(NotFittedError, match=re.escape(msg)):
transformer.transform(make_df(data_vartypes))