Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions docs/user_guide/encoding/MeanEncoder.rst
Original file line number Diff line number Diff line change
Expand Up @@ -364,6 +364,62 @@ After encoding the features we can use the data sets to train machine learning a
encoded variable. Hence, this encoding method is suitable for predictive modelling that
uses models that are sensitive to the size of the feature space.

With polars
~~~~~~~~~~~

:class:`MeanEncoder()` works the same way with a polars dataframe. Let's create a toy dataset:

.. code:: python

import polars as pl
from feature_engine.encoding import MeanEncoder

X = pl.DataFrame({
"city": ["London", "Manchester", "Liverpool", "London", "Manchester", "Liverpool"],
"price": [500, 300, 250, 520, 310, 260],
})
y = pl.Series("target", [1, 0, 0, 1, 0, 1])

Let's set up :class:`MeanEncoder()` to encode `city` with the target mean, and fit it to the data:

.. code:: python

encoder = MeanEncoder(variables=["city"])
encoder.fit(X, y)

encoder.encoder_dict_

We see the resulting mappings from category to target mean:

.. code:: python

{'city': {'London': 1.0, 'Liverpool': 0.5, 'Manchester': 0.0}}

Now let's transform the data:

.. code:: python

encoder.transform(X)

We obtain a polars dataframe with the categories in `city` replaced by the target mean:

.. code:: text

shape: (6, 2)
┌──────┬───────┐
│ city ┆ price │
│ --- ┆ --- │
│ f64 ┆ i64 │
╞══════╪═══════╡
│ 1.0 ┆ 500 │
│ 0.0 ┆ 300 │
│ 0.5 ┆ 250 │
│ 1.0 ┆ 520 │
│ 0.0 ┆ 310 │
│ 0.5 ┆ 260 │
└──────┴───────┘


Additional resources
--------------------

Expand Down
78 changes: 56 additions & 22 deletions feature_engine/encoding/mean_encoding.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,9 @@
# License: BSD 3 clause
from typing import List, Union

import pandas as pd
import narwhals as nw
import narwhals.dependencies as nwd
from narwhals.typing import IntoDataFrame, IntoSeries

from feature_engine._check_init_parameters.check_init_input_params import (
_check_return_empty_is_bool,
Expand Down Expand Up @@ -203,64 +205,96 @@ def __init__(
check_parameter_unseen(unseen, ["ignore", "raise", "encode"])
self.unseen = unseen

def fit(self, X: pd.DataFrame, y: pd.Series):
def fit(self, X: IntoDataFrame, y: IntoSeries):
"""
Learn the mean value of the target for each category of the variable.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features]
X: dataframe of shape = [n_samples, n_features]
The training input samples. Can be the entire dataframe, not just the
variables to be encoded.

y: pandas series
y: Series
The target.
"""

X, y = check_X_y(X, y)
nw_X, y = check_X_y(X, y)
variables_ = self._check_or_select_variables(X)
self._check_na(X, variables_)

self.encoder_dict_ = {}

y_prior = y.mean()
# pair y with X by position, so list, array and series targets all work
target_name = "__feature_engine_mean_target__"
if nwd.is_into_series(y):
y_nw = nw.from_native(y, series_only=True).alias(target_name)
else:
y_nw = nw.new_series(
name=target_name, values=y, backend=nw_X.implementation
)
nw_Xy = nw_X.with_columns(y_nw)

y_prior = y_nw.mean()

if self.unseen == "encode":
self._unseen = y_prior

if self.smoothing == "auto":
y_var = y.var(ddof=0)
for var in variables_:
if self.smoothing == "auto":
damping = y.groupby(X[var]).var(ddof=0) / y_var
else:
damping = self.smoothing
counts = X[var].value_counts()
counts.index = counts.index.infer_objects()
_lambda = counts / (counts + damping)
self.encoder_dict_[var] = (
_lambda * y.groupby(X[var], observed=False).mean()
+ (1.0 - _lambda) * y_prior
).to_dict()
y_var = y_nw.var(ddof=0)

# pandas is faster than narwhals.
if nwd.is_pandas_dataframe(X):
# pandas series with the index of X
y = nw_Xy[target_name].to_native()
for var in variables_:
if self.smoothing == "auto":
damping = y.groupby(X[var]).var(ddof=0) / y_var
else:
damping = self.smoothing
counts = X[var].value_counts()
counts.index = counts.index.infer_objects()
_lambda = counts / (counts + damping)
self.encoder_dict_[var] = (
_lambda * y.groupby(X[var], observed=False).mean()
+ (1.0 - _lambda) * y_prior
).to_dict()
else:
for var in variables_:
stats = nw_Xy.group_by(var, drop_null_keys=True).agg(
nw.col(target_name).mean().alias("__mean__"),
nw.col(target_name).len().alias("__count__"),
nw.col(target_name).var(ddof=0).alias("__var__"),
)
if self.smoothing == "auto":
damping = nw.col("__var__") / y_var
else:
damping = self.smoothing
_lambda = nw.col("__count__") / (nw.col("__count__") + damping)
encoding = _lambda * nw.col("__mean__") + (1.0 - _lambda) * y_prior
stats = stats.select(var, encoding.alias("__encoding__"))
self.encoder_dict_[var] = dict(
zip(stats[var].to_list(), stats["__encoding__"].to_list())
)

# assign underscore parameters at the end in case code above fails
self.variables_ = variables_
self._get_feature_names_in(X)
return self

def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame:
def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame:
"""Convert the encoded variable back to the original values.

Note that if unseen was set to 'encode', then this method is not implemented.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features].
X: dataframe of shape = [n_samples, n_features].
The transformed dataframe.

Returns
-------
X_tr: pandas dataframe of shape = [n_samples, n_features].
X_tr: dataframe of shape = [n_samples, n_features].
The un-transformed dataframe, with the categorical variables containing the
original values.
"""
Expand Down
Loading