Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 47 additions & 4 deletions docs/user_guide/discretisation/EqualWidthDiscretiser.rst
Original file line number Diff line number Diff line change
Expand Up @@ -48,9 +48,10 @@ potentially impact the model's performance in this scenario.
EqualWidthDiscretiser
---------------------

Feture-engine's :class:`EqualWidthDiscretiser()` applies equal width discretisation to numerical variables. It uses
the `pandas.cut()` function under the hood to find the interval limits and then sort the continuous variables into
the bins.
Feture-engine's :class:`EqualWidthDiscretiser()` applies equal width discretisation to numerical variables. It finds
the interval limits from each variable's minimum and maximum value, then sorts the continuous variables into the
bins. It works with pandas, polars, and any other dataframe library supported by
`narwhals <https://narwhals-dev.github.io/narwhals/>`_.

You can specify the variables to be discretised by passing their names in a list when you set up the transformer. Alternatively,
:class:`EqualWidthDiscretiser()` will automatically infer the data types and compute the interval limits for all numeric
Expand Down Expand Up @@ -271,7 +272,7 @@ If we want to output the intervals limits instead of integers, we can set `retur
.. code:: python

# Set up the discretisation transformer
disc = EqualFrequencyDiscretiser(
disc = EqualWidthDiscretiser(
bins=10,
variables=['LotArea','GrLivArea'],
return_boundaries=True)
Expand Down Expand Up @@ -301,6 +302,48 @@ While we can't use these
variables to train machine learning models, as opposed to the variables discretised into integers, they are very useful
in this format for data analysis, and we can use any feature-engine encoder for further processing.

With polars
~~~~~~~~~~~

:class:`EqualWidthDiscretiser()` works in the same way with a polars dataframe:

.. code:: python

import polars as pl
from feature_engine.discretisation import EqualWidthDiscretiser

df = pl.DataFrame({
"x": [10400, 3675, 8640, 11670, 10667, 6120, 9500, 14000, 7200, 5300],
})

disc = EqualWidthDiscretiser(bins=5)

print(disc.fit_transform(df))

The resulting values match those found with pandas:

.. code:: text

shape: (10, 1)
┌─────┐
│ x │
│ --- │
│ i64 │
╞═════╡
│ 3 │
│ 0 │
│ 2 │
│ 3 │
│ 3 │
│ 1 │
│ 2 │
│ 4 │
│ 1 │
│ 0 │
└─────┘

`return_object`, `return_boundaries`, and `binner_dict_` work identically to the pandas examples above.

See Also
--------

Expand Down
51 changes: 34 additions & 17 deletions feature_engine/discretisation/equal_width.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,9 @@

from typing import List, Optional, Union

import pandas as pd
import narwhals as nw
import numpy as np
from narwhals.typing import IntoDataFrame, IntoSeries

from feature_engine._check_init_parameters.check_init_input_params import (
_check_return_empty_is_bool,
Expand Down Expand Up @@ -164,14 +166,14 @@ def __init__(
self.return_empty = return_empty
self.bins = bins

def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None):
def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None):
"""
Learn the boundaries of the equal width intervals / bins for each
variable.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features]
X: dataframe of shape = [n_samples, n_features]
The training dataset. Can be the entire dataframe, not just the variables
to be transformed.
y: None
Expand All @@ -184,23 +186,38 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None):
# fit
binner_dict_ = {}

for var in variables_:
tmp, bins = pd.cut(
x=X[var],
bins=self.bins,
retbins=True,
duplicates="drop",
include_lowest=True,
)

# Prepend/Append infinities
bins = list(bins)
bins[0] = float("-inf")
bins[len(bins) - 1] = float("inf")
binner_dict_[var] = bins
if len(variables_) > 0:
# one narwhals call for every variable at once, instead of a
# get_column() round-trip per variable.
arr = nw.from_native(X, eager_only=True).select(variables_).to_numpy()
mins = arr.min(axis=0)
maxs = arr.max(axis=0)
for var, mn, mx in zip(variables_, mins, maxs):
binner_dict_[var] = self._equal_width_edges(mn, mx, self.bins)

self.binner_dict_ = binner_dict_
self.variables_ = variables_
self._get_feature_names_in(X)

return self

def _equal_width_edges(self, mn: float, mx: float, bins: int) -> List[float]:
"""Bin-edge computation matching pandas.cut(bins=int, duplicates="drop"):
widen a constant [mn, mx] by 0.1% so linspace still produces positive-
width bins, then collapse duplicate edges the same way. The outer edges
are then clipped to +-inf, same as the pre-migration code did to the
retbins output, so transform() never needs an out-of-range branch.
"""
if mn == mx:
mn = mn - 0.001 * abs(mn) if mn != 0 else -0.001
mx = mx + 0.001 * abs(mx) if mx != 0 else 0.001

edges = np.linspace(mn, mx, bins + 1)
unique_edges = np.unique(edges)
if len(unique_edges) < len(edges) and len(edges) != 2:
edges = unique_edges

edges_: List[float] = edges.tolist()
edges_[0] = float("-inf")
edges_[-1] = float("inf")
return edges_
163 changes: 109 additions & 54 deletions tests/test_discretisation/test_equal_width_discretiser.py
Original file line number Diff line number Diff line change
@@ -1,70 +1,125 @@
import re

import narwhals as nw
import numpy as np
import pandas as pd
import pytest
from sklearn.exceptions import NotFittedError

from feature_engine.discretisation import EqualWidthDiscretiser
from tests.backend_helpers import frame_to_dict

MSG_NA = (
"Some of the variables in the dataset contain NaN. Check and "
"remove those before using this transformer."
)


# init parameters
@pytest.mark.parametrize("bins", ["other", 1.5, None, [10]])
def test_error_when_bins_not_number(bins):
msg = f"bins must be an integer. Got {bins} instead."
with pytest.raises(ValueError, match=re.escape(msg)):
EqualWidthDiscretiser(bins=bins)


@pytest.mark.parametrize("return_object", ["other", 1, None])
def test_error_if_return_object_not_bool(return_object):
msg = f"return_object must be True or False. Got {return_object} instead."
with pytest.raises(ValueError, match=re.escape(msg)):
EqualWidthDiscretiser(return_object=return_object)


@pytest.mark.parametrize(
"bins, return_object, return_boundaries, precision",
[(10, False, False, 3), (5, True, False, 1), (2, False, True, 7)],
)
def test_init_param_assignment(bins, return_object, return_boundaries, precision):
transformer = EqualWidthDiscretiser(
bins=bins,
return_object=return_object,
return_boundaries=return_boundaries,
precision=precision,
)
assert transformer.bins == bins
assert transformer.return_object is return_object
assert transformer.return_boundaries is return_boundaries
assert transformer.precision == precision


# fit and transform

def _expected_bins_and_codes(values, n_bins):
# ground truth bin edges via pandas.cut, same widening/duplicates-drop
# rules the fit() replicates in plain numpy.
series = pd.Series(values)
_, bins = pd.cut(x=series, bins=n_bins, retbins=True, duplicates="drop")
bins[0] = float("-inf")
bins[len(bins) - 1] = float("inf")
codes = pd.cut(series, bins=list(bins), labels=False, include_lowest=True)
return bins, codes.tolist()


def test_automatically_find_variables_and_return_as_numeric(df_normal_dist):
# test case 1: automatically select variables, return_object=False
def test_automatically_find_variables_and_return_as_numeric(
make_df, data_normal_dist
):
transformer = EqualWidthDiscretiser(bins=10, variables=None, return_object=False)
X = transformer.fit_transform(df_normal_dist)

# fit parameters
_, bins = pd.cut(x=df_normal_dist["var"], bins=10, retbins=True, duplicates="drop")
bins[0] = float("-inf")
bins[len(bins) - 1] = float("inf")
X = transformer.fit_transform(make_df(data_normal_dist))

# transform output
X_t = [x for x in range(0, 10)]
val_counts = [18, 17, 16, 13, 11, 7, 7, 5, 5, 1]
bins, expected_codes = _expected_bins_and_codes(data_normal_dist["var"], 10)

# init params
assert transformer.bins == 10
assert transformer.variables is None
assert transformer.return_object is False
# fit params
assert transformer.variables_ == ["var"]
assert transformer.n_features_in_ == 1
# transform params
assert (transformer.binner_dict_["var"] == bins).all()
assert all(x for x in X["var"].unique() if x not in X_t)
# in equal width discretisation, intervals get different number of values
assert all(x for x in X["var"].value_counts() if x not in val_counts)
assert np.allclose(transformer.binner_dict_["var"], bins)
# transform params: same bin codes on both backends
assert isinstance(X, make_df)
assert frame_to_dict(X)["var"] == expected_codes


def test_automatically_find_variables_and_return_as_object(df_normal_dist):
def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist):
transformer = EqualWidthDiscretiser(bins=10, variables=None, return_object=True)
X = transformer.fit_transform(df_normal_dist)
assert X["var"].dtypes == "O"


def test_error_when_bins_not_number():
with pytest.raises(ValueError):
EqualWidthDiscretiser(bins="other")


def test_error_if_return_object_not_bool():
with pytest.raises(ValueError):
EqualWidthDiscretiser(return_object="other")


def test_error_if_input_df_contains_na_in_fit(df_na):
# test case 3: when dataset contains na, fit method
with pytest.raises(ValueError):
transformer = EqualWidthDiscretiser()
transformer.fit(df_na)


def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na):
# test case 4: when dataset contains na, transform method
with pytest.raises(ValueError):
transformer = EqualWidthDiscretiser()
transformer.fit(df_vartypes)
transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]])


def test_non_fitted_error(df_vartypes):
with pytest.raises(NotFittedError):
transformer = EqualWidthDiscretiser()
transformer.transform(df_vartypes)
X = transformer.fit_transform(make_df(data_normal_dist))
assert isinstance(X, make_df)
assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object


def test_constant_variable_produces_single_bin(make_df):
# a constant variable still fits, with every value in the same bin, as with
# pandas.cut(bins=10)
data = {"var": [5.0] * 10}
transformer = EqualWidthDiscretiser(bins=10)
X = transformer.fit_transform(make_df(data))

_, expected_codes = _expected_bins_and_codes(data["var"], 10)

assert transformer.binner_dict_["var"][0] == float("-inf")
assert transformer.binner_dict_["var"][-1] == float("inf")
assert isinstance(X, make_df)
assert frame_to_dict(X)["var"] == expected_codes


def test_error_if_input_df_contains_na_in_fit(make_df, data_na):
transformer = EqualWidthDiscretiser()
with pytest.raises(ValueError, match=re.escape(MSG_NA)):
transformer.fit(make_df(data_na))


def test_error_if_input_df_contains_na_in_transform(make_df, data_vartypes, data_na):
transform_data = make_df(
{k: data_na[k] for k in ["Name", "City", "Age", "Marks", "dob"]}
)
transformer = EqualWidthDiscretiser()
transformer.fit(make_df(data_vartypes))
with pytest.raises(ValueError, match=re.escape(MSG_NA)):
transformer.transform(transform_data)


def test_non_fitted_error(make_df, data_vartypes):
transformer = EqualWidthDiscretiser()
msg = (
"This EqualWidthDiscretiser instance is not fitted yet. Call 'fit' with "
"appropriate arguments before using this estimator."
)
with pytest.raises(NotFittedError, match=re.escape(msg)):
transformer.transform(make_df(data_vartypes))