From ba33564b0d6411778b6a57ee2d46e50cde31f715 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Wed, 26 Aug 2026 20:47:52 +0200 Subject: [PATCH 1/7] Migrate WoEEncoder to narwhals, add polars support fit() splits by backend: pandas keeps _calculate_woe()'s existing two-groupby implementation unchanged (it's directly unit-tested for that exact pandas-Series-with-category-index contract); polars/other narwhals backends use one group_by() instead of two, deriving the negative-class count as the complement of the positive-class count per category - benchmarked competitive with, and often faster than, pandas-native at 50k-100k rows. Zero-count-per-class fill_value handling preserved exactly. Bug fix: _check_fit_input() previously assumed y was always a pandas Series (y.nunique()/y.min()/y.max()), breaking on a numpy y (e.g. a plain list/array-like target, which sklearn's check_X_y machinery converts via column_or_1d). Wrapped numpy y into a narwhals Series aligned to X's backend; for pandas specifically, also had to line the wrapped Series up with X's actual index, since _calculate_woe()'s y.groupby(X[var]) aligns by index and a mismatched default RangeIndex silently drops every row instead of raising, leaving encoder_dict_ empty. Fixes test_encoders_when_x_pandas_y_numpy's WoEEncoder case (was failing on the unmigrated file, confirmed pre-existing). Verified: 44/44 own tests, full encoding suite 342 passed/16 failed (was 17 pre-existing on the narwhals-encoding-base baseline - one less here since this branch's own numpy-y bug is now fixed, rest confirmed unrelated), flake8 and mypy clean, sphinx -W build clean (only the pre-existing unrelated linkcode_resolve warning). Co-Authored-By: Claude Sonnet 5 --- docs/user_guide/encoding/WoEEncoder.rst | 35 +++ feature_engine/encoding/woe.py | 134 ++++++++++-- .../test_woe/test_woe_encoder.py | 204 ++++++++++-------- 3 files changed, 268 insertions(+), 105 deletions(-) diff --git a/docs/user_guide/encoding/WoEEncoder.rst b/docs/user_guide/encoding/WoEEncoder.rst index a25d83074..7b75e1a30 100644 --- a/docs/user_guide/encoding/WoEEncoder.rst +++ b/docs/user_guide/encoding/WoEEncoder.rst @@ -280,6 +280,41 @@ variable values: 686 -0.584173 female 22.000000 0 0 7.7250 -0.357528 0.012075 +With polars +~~~~~~~~~~~ + +:class:`WoEEncoder()` also works with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.encoding import WoEEncoder + + X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + y = pl.Series([0,1,1,1,0]) + + woe = WoEEncoder() + woe.fit(X, y) + woe.transform(X) + +We see the resulting dataframe below: + +.. code:: text + + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ + + WoE in categorical and numerical variables ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index bd1538a2c..06a95de99 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -3,8 +3,10 @@ from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -35,14 +37,37 @@ class WoE: - def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): + def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): """ Check that X is dataframe, and y a binary series with values 0 and 1. """ X, y = check_X_y(X, y) + if nwd.is_into_series(y): + y_nw = nw.from_native(y, series_only=True) + else: + # y is a numpy array here (e.g. list/array-like y input, which + # sklearn's check_X_y machinery converts to numpy via + # column_or_1d) - it has no .nunique()/.groupby(), so wrap it + # against X's backend to get one consistent narwhals Series. + is_pandas = nwd.is_pandas_dataframe(X) + y_nw = nw.new_series( + name="target", + values=y, + backend=nw.from_native(X, eager_only=True).implementation, + ) + if is_pandas is True: + # new_series() gives pandas a fresh default RangeIndex, but + # _calculate_woe()'s y.groupby(X[var]) aligns the two + # Series by index - a mismatch against X's own index + # silently drops every row instead of raising, leaving + # encoder_dict_ empty. Line it up with X's index. + native_y = y_nw.to_native() + native_y.index = X.index + y_nw = nw.from_native(native_y, series_only=True) + # check that y is binary - if y.nunique() != 2: + if y_nw.n_unique() != 2: raise ValueError( "This encoder is designed for binary classification. The target " "used has more than 2 unique values." @@ -50,14 +75,16 @@ def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): # if target does not have values 0 and 1, we need to remap, to be able to # compute the averages. - if y.min() != 0 or y.max() != 1: - y = pd.Series(np.where(y == y.min(), 0, 1)) - return X, y + y_min, y_max = y_nw.min(), y_nw.max() + if y_min != 0 or y_max != 1: + y_nw = (y_nw != y_min).cast(nw.Int64()).alias("target") + + return X, y_nw.to_native() def _calculate_woe( self, - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y: IntoSeries, variable: Union[str, int], fill_value: Union[float, None] = None, ): @@ -198,6 +225,28 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): 2 3 0.287682 3 4 -0.405465 4 5 -0.405465 + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import WoEEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + >>> y = pl.Series([0,1,1,1,0]) + >>> woe = WoEEncoder() + >>> woe.fit(X, y) + >>> woe.transform(X) + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ """ def __init__( @@ -218,17 +267,17 @@ def __init__( self.unseen = unseen self.fill_value = fill_value - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Learn the WoE. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the categorical variables. - y: pandas series. + y: Series. Target, must be binary. """ X, y = self._check_fit_input(X, y) @@ -238,12 +287,57 @@ def fit(self, X: pd.DataFrame, y: pd.Series): encoder_dict_ = {} vars_that_fail = [] - for var in variables_: - try: - _, _, woe = self._calculate_woe(X, y, var, self.fill_value) - encoder_dict_[var] = woe.to_dict() - except ValueError: - vars_that_fail.append(var) + # _calculate_woe() keeps its pandas-native two-groupby implementation + # (it's directly unit-tested for that exact pandas-Series-with- + # category-index return contract); polars and other narwhals + # backends compute the same ratio-then-log logic with a single + # group_by() instead - it derives the negative-class count as the + # complement of the positive-class count per category, so only one + # groupby is needed instead of two (benchmarked competitive with, + # and often faster than, pandas-native at 50k-100k rows). + is_pandas = nwd.is_pandas_dataframe(X) + + if is_pandas is True: + for var in variables_: + try: + _, _, woe = self._calculate_woe(X, y, var, self.fill_value) + encoder_dict_[var] = woe.to_dict() + except ValueError: + vars_that_fail.append(var) + else: + nw_X = nw.from_native(X, eager_only=True) + y_nw = nw.from_native(y, series_only=True) + target_name = "__feature_engine_woe_target__" + nw_Xy = nw_X.with_columns(y_nw.alias(target_name)) + + total_pos = y_nw.sum() + total_neg = len(y_nw) - total_pos + + for var in variables_: + grouped = ( + nw_Xy.group_by(var, drop_null_keys=True) + .agg( + nw.col(target_name).sum().alias("__pos_n__"), + nw.len().alias("__n__"), + ) + .sort(var) + ) + categories = grouped.get_column(var).to_list() + pos = (grouped.get_column("__pos_n__") / total_pos).to_numpy() + neg = ( + (grouped.get_column("__n__") - grouped.get_column("__pos_n__")) + / total_neg + ).to_numpy() + + if (pos == 0).any() or (neg == 0).any(): + if self.fill_value is None: + vars_that_fail.append(var) + continue + pos = np.where(pos == 0, self.fill_value, pos) + neg = np.where(neg == 0, self.fill_value, neg) + + woe = np.log(pos / neg) + encoder_dict_[var] = dict(zip(categories, woe)) if len(vars_that_fail) > 0: vars_that_fail_str = ( @@ -263,17 +357,17 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Replace categories with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The dataset to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features]. + X_new: dataframe of shape = [n_samples, n_features]. The dataframe containing the categories replaced by numbers. """ diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index a38caa6fa..ab05eb289 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -1,7 +1,10 @@ import math +import re +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError @@ -53,17 +56,55 @@ 0.8472978603872037, ] - -def test_automatically_select_variables(df_enc): +DF_ENC = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +} + +DF_ENC_NUMERIC = { + "var_A": [1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3], + "var_B": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3], + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +} + +DF_ENC_RARE = { + "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +} + +# None (not np.nan) is what both pandas and polars accept as a missing +# value inside a string column literal. +DF_ENC_NA = { + "var_A": [None] + ["B"] * 8 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +} + + +def _none_to_nan(values): + # Missing values print as None for polars, NaN for pandas float columns + # - both mean "missing" here, so normalize both sides before comparing. + return [np.nan if v is None else v for v in values] + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert _none_to_nan(result[col]) == pytest.approx( + _none_to_nan(values), abs=abs_tol, nan_ok=True + ) + + +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_select_variables(make_df): + df_enc = make_df(DF_ENC) encoder = WoEEncoder(variables=None) encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) X = encoder.transform(df_enc[["var_A", "var_B"]]) - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - assert encoder.encoder_dict_ == { "var_A": { "A": 0.15415067982725836, @@ -76,19 +117,16 @@ def test_automatically_select_variables(df_enc): "C": 0.8472978603872037, }, } - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B}) -def test_user_passes_variables(df_enc): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_user_passes_variables(make_df): + df_enc = make_df(DF_ENC) encoder = WoEEncoder(variables=["var_A", "var_B"]) encoder.fit(df_enc, df_enc["target"]) X = encoder.transform(df_enc) - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - assert encoder.encoder_dict_ == { "var_A": { "A": 0.15415067982725836, @@ -101,7 +139,9 @@ def test_user_passes_variables(df_enc): "C": 0.8472978603872037, }, } - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal( + X, {"var_A": VAR_A, "var_B": VAR_B, "target": DF_ENC["target"]} + ) _targets = [ @@ -111,18 +151,16 @@ def test_user_passes_variables(df_enc): ] +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("target", _targets) -def test_when_target_class_not_0_1(df_enc, target): +def test_when_target_class_not_0_1(make_df, target): + data = dict(DF_ENC) + data["target"] = target + df_enc = make_df(data) encoder = WoEEncoder(variables=["var_A", "var_B"]) - df_enc["target"] = target encoder.fit(df_enc, df_enc["target"]) X = encoder.transform(df_enc) - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - assert encoder.encoder_dict_ == { "var_A": { "A": 0.15415067982725836, @@ -135,10 +173,13 @@ def test_when_target_class_not_0_1(df_enc, target): "C": 0.8472978603872037, }, } - pd.testing.assert_frame_equal(X, transf_df) + assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B, "target": target}) -def test_warn_if_transform_df_contains_categories_not_seen_in_fit(df_enc, df_enc_rare): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_warn_if_transform_df_contains_categories_not_seen_in_fit(make_df): + df_enc = make_df(DF_ENC) + df_enc_rare = make_df(DF_ENC_RARE) # test case 3: when dataset to be transformed contains categories not present # in training dataset msg = "During the encoding, NaN values were introduced in the feature(s) var_A." @@ -156,16 +197,14 @@ def test_warn_if_transform_df_contains_categories_not_seen_in_fit(df_enc, df_enc assert any(r.message.args[0] == msg for r in record) # check for error when rare_labels equals 'raise' - with pytest.raises(ValueError) as record: - encoder = WoEEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder = WoEEncoder(unseen="raise") + encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + with pytest.raises(ValueError, match=re.escape(msg)): encoder.transform(df_enc_rare[["var_A", "var_B"]]) - # check that the error message matches - assert str(record.value) == msg - -def test_error_if_target_not_binary(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_target_not_binary(make_df): # test case 4: the target is not binary encoder = WoEEncoder(variables=None) with pytest.raises(ValueError): @@ -174,107 +213,101 @@ def test_error_if_target_not_binary(): "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder.fit(df[["var_A", "var_B"]], df["target"]) -def test_error_if_denominator_probability_is_zero_1_var(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_denominator_probability_is_zero_1_var(make_df): df = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - msg = ( "During the WoE calculation, some of the categories in the " "following features contained 0 in the denominator or numerator, " "and hence the WoE can't be calculated: var_A." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.fit(df[["var_A", "var_B"]], df["target"]) df = { "var_A": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_B": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - msg = ( "During the WoE calculation, some of the categories in the " "following features contained 0 in the denominator or numerator, " "and hence the WoE can't be calculated: var_B." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.fit(df[["var_A", "var_B"]], df["target"]) -def test_error_if_denominator_probability_is_zero_2_vars(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_denominator_probability_is_zero_2_vars(make_df): df = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) - msg = ( "During the WoE calculation, some of the categories in the " "following features contained 0 in the denominator or numerator, " "and hence the WoE can't be calculated: var_A, var_C." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.fit(df, df["target"]) -def test_error_if_numerator_probability_is_zero(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_numerator_probability_is_zero(make_df): df = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) - msg = ( "During the WoE calculation, some of the categories in the " "following features contained 0 in the denominator or numerator, " "and hence the WoE can't be calculated: var_A, var_C." ) - assert str(record.value) == msg - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) + with pytest.raises(ValueError, match=msg): + encoder.fit(df, df["target"]) msg = ( "During the WoE calculation, some of the categories in the " "following features contained 0 in the denominator or numerator, " "and hence the WoE can't be calculated: var_A." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.fit(df[["var_A", "var_B"]], df["target"]) -def test_fill_value(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fill_value(make_df): df = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = pd.DataFrame(df) + df = make_df(df) encoder = WoEEncoder(variables=None, fill_value=1) encoder.fit(df, df["target"]) woe_exp_a = { @@ -320,43 +353,42 @@ def test_assigns_fill_value_at_init(fill_value): assert encoder.fill_value == fill_value -def test_error_if_contains_na_in_fit(df_enc_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_contains_na_in_fit(make_df): # test case 9: when dataset contains na, fit method + df_enc_na = make_df(DF_ENC_NA) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) - msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) -def test_error_if_df_contains_na_in_transform(df_enc, df_enc_na): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_df_contains_na_in_transform(make_df): # test case 10: when dataset contains na, transform method} + df_enc = make_df(DF_ENC) + df_enc_na = make_df(DF_ENC_NA) encoder = WoEEncoder(variables=None) encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na[["var_A", "var_B"]]) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=msg): + encoder.transform(df_enc_na[["var_A", "var_B"]]) -def test_on_numerical_variables(df_enc_numeric): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_on_numerical_variables(make_df): # ignore_format=True + df_enc_numeric = make_df(DF_ENC_NUMERIC) encoder = WoEEncoder(variables=None, ignore_format=True) encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) - # transformed dataframe - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - # init params assert encoder.variables is None # fit params @@ -375,16 +407,17 @@ def test_on_numerical_variables(df_enc_numeric): } assert encoder.n_features_in_ == 2 # transform params - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B}) -def test_variables_cast_as_category(df_enc_category_dtypes): - df = df_enc_category_dtypes.copy() +def test_variables_cast_as_category(): + # pandas Categorical dtype has no direct polars equivalent. + df = pd.DataFrame(DF_ENC) + df[["var_A", "var_B"]] = df[["var_A", "var_B"]].astype("category") encoder = WoEEncoder(variables=None) encoder.fit(df[["var_A", "var_B"]], df["target"]) X = encoder.transform(df[["var_A", "var_B"]]) - # transformed dataframe transf_df = df.copy() transf_df["var_A"] = VAR_A transf_df["var_B"] = VAR_B @@ -401,19 +434,20 @@ def test_error_if_rare_labels_not_permitted_value(errors): WoEEncoder(unseen=errors) -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) enc = WoEEncoder() # Test when fit is not called prior to transform. with pytest.raises(NotFittedError): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = np.nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) with pytest.raises(ValueError): - enc.fit(df1, pd.Series([0, 1, 0, 1, 1, 0])) + enc.fit(df1_na, make_df({"target": [0, 1, 0, 1, 1, 0]})["target"]) # Test when fit is not called prior to transform. with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + enc.inverse_transform(df1_na) From f132b5f1bb96a7c5919e98eb7cb6bdf2c9c82712 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Mon, 31 Aug 2026 00:48:13 +0200 Subject: [PATCH 2/7] Adapt WoEEncoder to narwhals-returning check_X check_X_y now returns a narwhals frame. In _check_fit_input, bind that to nw_X and keep the original native X: the nwd.is_pandas_dataframe(X) check, the native_y.index = X.index alignment and the returned X all need native input, and fit()'s pandas _calculate_woe fast path and nwd checks are then unchanged (X stays native so no rehydration is needed). Take the y-series backend from nw_X.implementation instead of re-wrapping X. In transform(), bind _check_transform_input_and_state to nw_X, keep native X for _check_contains_na, and pass nw_X to _encode (which now expects narwhals). Co-Authored-By: Claude Sonnet 5 --- feature_engine/encoding/woe.py | 15 ++++++--------- 1 file changed, 6 insertions(+), 9 deletions(-) diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index 06a95de99..4e203b62a 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -41,7 +41,7 @@ def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): """ Check that X is dataframe, and y a binary series with values 0 and 1. """ - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) if nwd.is_into_series(y): y_nw = nw.from_native(y, series_only=True) @@ -50,13 +50,12 @@ def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): # sklearn's check_X_y machinery converts to numpy via # column_or_1d) - it has no .nunique()/.groupby(), so wrap it # against X's backend to get one consistent narwhals Series. - is_pandas = nwd.is_pandas_dataframe(X) y_nw = nw.new_series( name="target", values=y, - backend=nw.from_native(X, eager_only=True).implementation, + backend=nw_X.implementation, ) - if is_pandas is True: + if nwd.is_pandas_dataframe(X): # new_series() gives pandas a fresh default RangeIndex, but # _calculate_woe()'s y.groupby(X[var]) aligns the two # Series by index - a mismatch against X's own index @@ -295,9 +294,7 @@ def fit(self, X: IntoDataFrame, y: IntoSeries): # complement of the positive-class count per category, so only one # groupby is needed instead of two (benchmarked competitive with, # and often faster than, pandas-native at 50k-100k rows). - is_pandas = nwd.is_pandas_dataframe(X) - - if is_pandas is True: + if nwd.is_pandas_dataframe(X): for var in variables_: try: _, _, woe = self._calculate_woe(X, y, var, self.fill_value) @@ -371,9 +368,9 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: The dataframe containing the categories replaced by numbers. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) _check_contains_na(X, self.variables_) - X = self._encode(X) + X = self._encode(nw_X) return X def _more_tags(self): From fc25eb3e814498ced4d955a4a20ab5e4aaeb79bc Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Mon, 14 Sep 2026 22:25:59 +0200 Subject: [PATCH 3/7] Use shared backend test fixtures and helpers in WoEEncoder tests Replace the file-local data dicts and assert_df_equal/_none_to_nan helpers with the shared test structure: make_df and data_enc* fixtures, y built with make_series on the backend under test, isinstance(X, make_df) plus to_dict() checks, and pytest.raises/warns(match=re.escape(msg)). Add a test passing the target as a list and as a numpy array, which take a different code path than a Series. Co-Authored-By: Claude Opus 5 --- .../test_woe/test_woe_encoder.py | 418 +++++++----------- 1 file changed, 159 insertions(+), 259 deletions(-) diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index ab05eb289..8588c31b5 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -1,147 +1,90 @@ import math import re -import narwhals as nw import numpy as np import pandas as pd -import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.encoding import WoEEncoder +from tests.backend_helpers import make_series, to_dict -VAR_A = [ - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] - -VAR_B = [ - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] - -DF_ENC = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +WOE_A = { + "A": 0.15415067982725836, + "B": -0.5389965007326869, + "C": 0.8472978603872037, } - -DF_ENC_NUMERIC = { - "var_A": [1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3], - "var_B": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3], - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], +WOE_B = { + "A": -0.5389965007326869, + "B": 0.15415067982725836, + "C": 0.8472978603872037, } +VAR_A = [WOE_A["A"]] * 6 + [WOE_A["B"]] * 10 + [WOE_A["C"]] * 4 +VAR_B = [WOE_B["A"]] * 10 + [WOE_B["B"]] * 6 + [WOE_B["C"]] * 4 -DF_ENC_RARE = { - "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], -} +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -# None (not np.nan) is what both pandas and polars accept as a missing -# value inside a string column literal. -DF_ENC_NA = { - "var_A": [None] + ["B"] * 8 + ["A"] * 6 + ["C"] * 4 + ["D"] * 1, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], -} +def _msg_zero_division(features): + return ( + "During the WoE calculation, some of the categories in the " + "following features contained 0 in the denominator or numerator, " + f"and hence the WoE can't be calculated: {features}." + ) -def _none_to_nan(values): - # Missing values print as None for polars, NaN for pandas float columns - # - both mean "missing" here, so normalize both sides before comparing. - return [np.nan if v is None else v for v in values] +def test_automatically_select_variables(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) -def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - assert list(result.keys()) == list(expected.keys()) - for col, values in expected.items(): - assert _none_to_nan(result[col]) == pytest.approx( - _none_to_nan(values), abs=abs_tol, nan_ok=True - ) + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_automatically_select_variables(make_df): - df_enc = make_df(DF_ENC) - encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), } - assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B}) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_user_passes_variables(make_df): - df_enc = make_df(DF_ENC) - encoder = WoEEncoder(variables=["var_A", "var_B"]) - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) +def test_user_passes_variables(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": data_enc["target"], } - assert_df_equal( - X, {"var_A": VAR_A, "var_B": VAR_B, "target": DF_ENC["target"]} - ) _targets = [ @@ -151,165 +94,132 @@ def test_user_passes_variables(make_df): ] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) @pytest.mark.parametrize("target", _targets) -def test_when_target_class_not_0_1(make_df, target): - data = dict(DF_ENC) +def test_when_target_class_not_0_1(make_df, data_enc, target): + data = dict(data_enc) data["target"] = target - df_enc = make_df(data) - encoder = WoEEncoder(variables=["var_A", "var_B"]) - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) + X = make_df(data) + y = make_series(make_df, target) - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": target, } - assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B, "target": target}) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_warn_if_transform_df_contains_categories_not_seen_in_fit(make_df): - df_enc = make_df(DF_ENC) - df_enc_rare = make_df(DF_ENC_RARE) +def test_warn_if_transform_df_contains_categories_not_seen_in_fit( + make_df, data_enc, data_enc_rare +): # test case 3: when dataset to be transformed contains categories not present # in training dataset + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_rare = make_df(data_enc_rare)[["var_A", "var_B"]] msg = "During the encoding, NaN values were introduced in the feature(s) var_A." - # check for error when rare_labels equals 'raise' - with pytest.warns(UserWarning) as record: - encoder = WoEEncoder(unseen="ignore") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) - - # check that at least one warning was raised (Pandas 3 may emit additional - # deprecation warnings) - assert len(record) >= 1 - # check that the message matches - assert any(r.message.args[0] == msg for r in record) + # check for warning when unseen equals 'ignore' + encoder = WoEEncoder(unseen="ignore") + encoder.fit(X, y) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(X_rare) - # check for error when rare_labels equals 'raise' + # check for error when unseen equals 'raise' encoder = WoEEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) + encoder.fit(X, y) with pytest.raises(ValueError, match=re.escape(msg)): - encoder.transform(df_enc_rare[["var_A", "var_B"]]) + encoder.transform(X_rare) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_target_not_binary(make_df): # test case 4: the target is not binary + data = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + } + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) + encoder = WoEEncoder(variables=None) with pytest.raises(ValueError): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = make_df(df) - encoder.fit(df[["var_A", "var_B"]], df["target"]) + encoder.fit(X, y) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_denominator_probability_is_zero_1_var(make_df): - df = { + data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = make_df(df) encoder = WoEEncoder(variables=None) + with pytest.raises(ValueError, match=re.escape(_msg_zero_division("var_A"))): + encoder.fit( + make_df(data)[["var_A", "var_B"]], make_series(make_df, data["target"]) + ) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - df = { + data = { "var_A": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_B": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = make_df(df) encoder = WoEEncoder(variables=None) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_B." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df[["var_A", "var_B"]], df["target"]) + with pytest.raises(ValueError, match=re.escape(_msg_zero_division("var_B"))): + encoder.fit( + make_df(data)[["var_A", "var_B"]], make_series(make_df, data["target"]) + ) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_denominator_probability_is_zero_2_vars(make_df): - df = { + data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = make_df(df) encoder = WoEEncoder(variables=None) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df, df["target"]) + msg = _msg_zero_division("var_A, var_C") + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(make_df(data), make_series(make_df, data["target"])) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_numerator_probability_is_zero(make_df): - df = { + data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "target": [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = make_df(df) + X = make_df(data) + y = make_series(make_df, data["target"]) encoder = WoEEncoder(variables=None) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df, df["target"]) + msg = _msg_zero_division("var_A, var_C") + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df[["var_A", "var_B"]], df["target"]) + msg = _msg_zero_division("var_A") + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X[["var_A", "var_B"]], y) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_fill_value(make_df): - df = { + data = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = make_df(df) + X = make_df(data) + y = make_series(make_df, data["target"]) + encoder = WoEEncoder(variables=None, fill_value=1) - encoder.fit(df, df["target"]) + encoder.fit(X, y) woe_exp_a = { "A": -0.6337237600891445, "B": -0.07410797215372196, @@ -328,7 +238,7 @@ def test_fill_value(make_df): assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) encoder = WoEEncoder(variables=None, fill_value=10) - encoder.fit(df, df["target"]) + encoder.fit(X, y) woe_exp_a = { "A": -0.6337237600891445, "B": -0.07410797215372196, @@ -353,67 +263,57 @@ def test_assigns_fill_value_at_init(fill_value): assert encoder.fill_value == fill_value -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_error_if_contains_na_in_fit(make_df): +def test_error_if_contains_na_in_fit(make_df, data_enc_na): # test case 9: when dataset contains na, fit method - df_enc_na = make_df(DF_ENC_NA) + X = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_na["target"]) + encoder = WoEEncoder(variables=None) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - with pytest.raises(ValueError, match=msg): - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(X, y) + +def test_error_if_df_contains_na_in_transform(make_df, data_enc, data_enc_na): + # test case 10: when dataset contains na, transform method + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_na = make_df(data_enc_na)[["var_A", "var_B"]] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_error_if_df_contains_na_in_transform(make_df): - # test case 10: when dataset contains na, transform method} - df_enc = make_df(DF_ENC) - df_enc_na = make_df(DF_ENC_NA) encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - with pytest.raises(ValueError, match=msg): - encoder.transform(df_enc_na[["var_A", "var_B"]]) + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.transform(X_na) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_on_numerical_variables(make_df): +def test_on_numerical_variables(make_df, data_enc_numeric): # ignore_format=True - df_enc_numeric = make_df(DF_ENC_NUMERIC) + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) + encoder = WoEEncoder(variables=None, ignore_format=True) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + encoder.fit(X, y) + Xt = encoder.transform(X) # init params assert encoder.variables is None # fit params assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { - "var_A": { - 1: 0.15415067982725836, - 2: -0.5389965007326869, - 3: 0.8472978603872037, - }, - "var_B": { - 1: -0.5389965007326869, - 2: 0.15415067982725836, - 3: 0.8472978603872037, - }, + "var_A": {1: WOE_A["A"], 2: WOE_A["B"], 3: WOE_A["C"]}, + "var_B": {1: WOE_B["A"], 2: WOE_B["B"], 3: WOE_B["C"]}, } assert encoder.n_features_in_ == 2 # transform params - assert_df_equal(X, {"var_A": VAR_A, "var_B": VAR_B}) + assert isinstance(Xt, make_df) + assert to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } -def test_variables_cast_as_category(): +def test_variables_cast_as_category(df_enc_category_dtypes): # pandas Categorical dtype has no direct polars equivalent. - df = pd.DataFrame(DF_ENC) - df[["var_A", "var_B"]] = df[["var_A", "var_B"]].astype("category") + df = df_enc_category_dtypes.copy() encoder = WoEEncoder(variables=None) encoder.fit(df[["var_A", "var_B"]], df["target"]) X = encoder.transform(df[["var_A", "var_B"]]) @@ -434,9 +334,9 @@ def test_error_if_rare_labels_not_permitted_value(errors): WoEEncoder(unseen=errors) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_inverse_transform_raises_non_fitted_error(make_df): df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 1, 0, 1, 1, 0]) enc = WoEEncoder() # Test when fit is not called prior to transform. @@ -446,7 +346,7 @@ def test_inverse_transform_raises_non_fitted_error(make_df): df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) with pytest.raises(ValueError): - enc.fit(df1_na, make_df({"target": [0, 1, 0, 1, 1, 0]})["target"]) + enc.fit(df1_na, y) # Test when fit is not called prior to transform. with pytest.raises(NotFittedError): From 1a60ca8cd2f4f298537bac9d79dc672ed051b313 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 12:01:47 +0200 Subject: [PATCH 4/7] Use frame_to_dict after the shared helper rename in #1045 Co-Authored-By: Claude Opus 5 --- tests/test_encoding/test_woe/test_woe_encoder.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index 8588c31b5..329850860 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -7,7 +7,7 @@ from sklearn.exceptions import NotFittedError from feature_engine.encoding import WoEEncoder -from tests.backend_helpers import make_series, to_dict +from tests.backend_helpers import make_series, frame_to_dict WOE_A = { "A": 0.15415067982725836, @@ -46,7 +46,7 @@ def test_automatically_select_variables(make_df, data_enc): assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} assert isinstance(Xt, make_df) - assert to_dict(Xt) == { + assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), "var_B": pytest.approx(VAR_B), } @@ -64,7 +64,7 @@ def test_target_as_list_or_array(make_df, data_enc, to_target): assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} assert isinstance(Xt, make_df) - assert to_dict(Xt) == { + assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), "var_B": pytest.approx(VAR_B), } @@ -80,7 +80,7 @@ def test_user_passes_variables(make_df, data_enc): assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} assert isinstance(Xt, make_df) - assert to_dict(Xt) == { + assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), "var_B": pytest.approx(VAR_B), "target": data_enc["target"], @@ -107,7 +107,7 @@ def test_when_target_class_not_0_1(make_df, data_enc, target): assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} assert isinstance(Xt, make_df) - assert to_dict(Xt) == { + assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), "var_B": pytest.approx(VAR_B), "target": target, @@ -305,7 +305,7 @@ def test_on_numerical_variables(make_df, data_enc_numeric): assert encoder.n_features_in_ == 2 # transform params assert isinstance(Xt, make_df) - assert to_dict(Xt) == { + assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), "var_B": pytest.approx(VAR_B), } From 4acf87417e4242e9d9daf15663c5f9b34c7d0825 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 13:33:56 +0200 Subject: [PATCH 5/7] Use add_target_to_X in WoEEncoder, group init tests, match errors Co-Authored-By: Claude Opus 5 --- feature_engine/encoding/woe.py | 59 ++++----------- .../test_encoding/test_woe/test_woe_class.py | 9 ++- .../test_woe/test_woe_encoder.py | 74 ++++++++++++------- 3 files changed, 70 insertions(+), 72 deletions(-) diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index 4e203b62a..e166f6346 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -28,7 +28,11 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import _check_contains_na, check_X_y -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) from feature_engine.encoding.base_encoder import ( CategoricalInitMixin, CategoricalMethodsMixin, @@ -42,28 +46,8 @@ def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): Check that X is dataframe, and y a binary series with values 0 and 1. """ nw_X, y = check_X_y(X, y) - - if nwd.is_into_series(y): - y_nw = nw.from_native(y, series_only=True) - else: - # y is a numpy array here (e.g. list/array-like y input, which - # sklearn's check_X_y machinery converts to numpy via - # column_or_1d) - it has no .nunique()/.groupby(), so wrap it - # against X's backend to get one consistent narwhals Series. - y_nw = nw.new_series( - name="target", - values=y, - backend=nw_X.implementation, - ) - if nwd.is_pandas_dataframe(X): - # new_series() gives pandas a fresh default RangeIndex, but - # _calculate_woe()'s y.groupby(X[var]) aligns the two - # Series by index - a mismatch against X's own index - # silently drops every row instead of raising, leaving - # encoder_dict_ empty. Line it up with X's index. - native_y = y_nw.to_native() - native_y.index = X.index - y_nw = nw.from_native(native_y, series_only=True) + # with pandas, y takes the index of X + y_nw = add_target_to_X(nw_X, y)[TARGET_NAME] # check that y is binary if y_nw.n_unique() != 2: @@ -286,14 +270,8 @@ def fit(self, X: IntoDataFrame, y: IntoSeries): encoder_dict_ = {} vars_that_fail = [] - # _calculate_woe() keeps its pandas-native two-groupby implementation - # (it's directly unit-tested for that exact pandas-Series-with- - # category-index return contract); polars and other narwhals - # backends compute the same ratio-then-log logic with a single - # group_by() instead - it derives the negative-class count as the - # complement of the positive-class count per category, so only one - # groupby is needed instead of two (benchmarked competitive with, - # and often faster than, pandas-native at 50k-100k rows). + # pandas uses _calculate_woe(); other backends count the negative class + # as total minus positive, so they need one group_by instead of two if nwd.is_pandas_dataframe(X): for var in variables_: try: @@ -302,19 +280,16 @@ def fit(self, X: IntoDataFrame, y: IntoSeries): except ValueError: vars_that_fail.append(var) else: - nw_X = nw.from_native(X, eager_only=True) - y_nw = nw.from_native(y, series_only=True) - target_name = "__feature_engine_woe_target__" - nw_Xy = nw_X.with_columns(y_nw.alias(target_name)) + nw_Xy = add_target_to_X(nw.from_native(X, eager_only=True), y) - total_pos = y_nw.sum() - total_neg = len(y_nw) - total_pos + total_pos = nw_Xy[TARGET_NAME].sum() + total_neg = len(nw_Xy) - total_pos for var in variables_: grouped = ( nw_Xy.group_by(var, drop_null_keys=True) .agg( - nw.col(target_name).sum().alias("__pos_n__"), + nw.col(TARGET_NAME).sum().alias("__pos_n__"), nw.len().alias("__n__"), ) .sort(var) @@ -377,12 +352,8 @@ def _more_tags(self): tags_dict = _return_tags() tags_dict["variables"] = "categorical" tags_dict["requires_y"] = True - # in the current format, the tests are performed using continuous np.arrays - # this means that when we encode some of the values, the denominator is 0 - # and this the transformer raises an error, and the test fails. - # For this reason, most sklearn tests will fail. And it has nothing to - # do with the class not being compatible, it is just that the inputs passed - # are not suitable + # sklearn tests pass continuous arrays, which give zero denominators and + # make this transformer raise, so they are skipped tags_dict["_skip_test"] = True return tags_dict diff --git a/tests/test_encoding/test_woe/test_woe_class.py b/tests/test_encoding/test_woe/test_woe_class.py index f26253786..9fcfa4169 100644 --- a/tests/test_encoding/test_woe/test_woe_class.py +++ b/tests/test_encoding/test_woe/test_woe_class.py @@ -1,3 +1,5 @@ +import re + import numpy as np import pandas as pd import pytest @@ -25,8 +27,11 @@ def test_woe_error(): } df = pd.DataFrame(df) woe_class = WoE() - - with pytest.raises(ValueError): + msg = ( + "The proportion of one of the classes for a category in variable var_A " + "is zero, and log of zero is not defined" + ) + with pytest.raises(ValueError, match=re.escape(msg)): woe_class._calculate_woe(df, df["target"], "var_A") diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index 329850860..35a6cdb2a 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -36,6 +36,42 @@ def _msg_zero_division(features): ) +# init parameters +@pytest.mark.parametrize("fill_value", ["hola", [10], (1,)]) +def test_error_if_fill_value_not_allowed(fill_value): + msg = f"fill_value takes None, integer or float. Got {fill_value} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WoEEncoder(fill_value=fill_value) + + +@pytest.mark.parametrize( + "unseen", ["empanada", "encode", False, 1, None, ("raise", "ignore"), ["ignore"]] +) +def test_error_if_unseen_not_permitted_value(unseen): + msg = f"Parameter `unseen` takes only values ignore, raise. Got {unseen} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WoEEncoder(unseen=unseen) + + +@pytest.mark.parametrize( + "ignore_format, unseen, fill_value", + [ + (False, "ignore", None), + (True, "raise", 0.5), + (False, "raise", 10), + (True, "ignore", 0), + ], +) +def test_init_param_assignment(ignore_format, unseen, fill_value): + encoder = WoEEncoder( + ignore_format=ignore_format, unseen=unseen, fill_value=fill_value + ) + assert encoder.ignore_format is ignore_format + assert encoder.unseen == unseen + assert encoder.fill_value == fill_value + + +# fit and transform def test_automatically_select_variables(make_df, data_enc): X = make_df(data_enc)[["var_A", "var_B"]] y = make_series(make_df, data_enc["target"]) @@ -148,7 +184,11 @@ def test_error_if_target_not_binary(make_df): y = make_series(make_df, data["target"]) encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError): + msg = ( + "This encoder is designed for binary classification. The target " + "used has more than 2 unique values." + ) + with pytest.raises(ValueError, match=re.escape(msg)): encoder.fit(X, y) @@ -251,18 +291,6 @@ def test_fill_value(make_df): assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) -@pytest.mark.parametrize("fill_value", ["hola", [10]]) -def test_error_if_fill_value_not_allowed(fill_value): - with pytest.raises(ValueError): - WoEEncoder(fill_value=fill_value) - - -@pytest.mark.parametrize("fill_value", [0, 1, 10, 0.5, 0.002, None]) -def test_assigns_fill_value_at_init(fill_value): - encoder = WoEEncoder(fill_value=fill_value) - assert encoder.fill_value == fill_value - - def test_error_if_contains_na_in_fit(make_df, data_enc_na): # test case 9: when dataset contains na, fit method X = make_df(data_enc_na)[["var_A", "var_B"]] @@ -294,8 +322,6 @@ def test_on_numerical_variables(make_df, data_enc_numeric): encoder.fit(X, y) Xt = encoder.transform(X) - # init params - assert encoder.variables is None # fit params assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { @@ -326,28 +352,24 @@ def test_variables_cast_as_category(df_enc_category_dtypes): assert X["var_A"].dtypes.name == "float64" -@pytest.mark.parametrize( - "errors", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] -) -def test_error_if_rare_labels_not_permitted_value(errors): - with pytest.raises(ValueError): - WoEEncoder(unseen=errors) - - def test_inverse_transform_raises_non_fitted_error(make_df): df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) y = make_series(make_df, [0, 1, 0, 1, 1, 0]) enc = WoEEncoder() + msg = ( + "This WoEEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=re.escape(MSG_NA)): enc.fit(df1_na, y) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1_na) From f88de400db3701d6cc82c93872975f48e3b5d2e7 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 14:24:36 +0200 Subject: [PATCH 6/7] Compute WoE with narwhals in _calculate_woe, shared by WoEEncoder and SelectByInformationValue Co-Authored-By: Claude Opus 5 --- feature_engine/encoding/woe.py | 100 ++++++++---------- feature_engine/selection/information_value.py | 8 +- .../test_encoding/test_woe/test_woe_class.py | 91 +++++++--------- .../test_woe/test_woe_encoder.py | 10 ++ 4 files changed, 100 insertions(+), 109 deletions(-) diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index e166f6346..95b6c9589 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -4,8 +4,6 @@ from typing import List, Union import narwhals as nw -import narwhals.dependencies as nwd -import numpy as np from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( @@ -71,27 +69,48 @@ def _calculate_woe( variable: Union[str, int], fill_value: Union[float, None] = None, ): - total_pos = y.sum() - inverse_y = y.ne(1).copy() - total_neg = inverse_y.sum() - - pos = y.groupby(X[variable], observed=False).sum() / total_pos - neg = inverse_y.groupby(X[variable], observed=False).sum() / total_neg + """ + Return a narwhals dataframe with one row per category of the variable and the + columns __category__, __pos__ and __neg__, the fraction of positive and + negative cases, and __woe__, the weight of evidence. + """ + # narwhals expressions need string column names, pandas allows integers + col = nw.from_native(X, eager_only=True).get_column(variable) + nw_Xy = add_target_to_X(col.alias("__category__").to_frame(), y) + total_pos = nw_Xy[TARGET_NAME].sum() + total_neg = len(nw_Xy) - total_pos + + stats = ( + nw_Xy.group_by("__category__", drop_null_keys=True) + .agg(nw.col(TARGET_NAME).sum().alias("__pos__"), nw.len().alias("__n__")) + .sort("__category__") + .select( + "__category__", + (nw.col("__pos__") / total_pos).alias("__pos__"), + ((nw.col("__n__") - nw.col("__pos__")) / total_neg).alias("__neg__"), + ) + ) - if not (pos[:] == 0).sum() == 0 or not (neg[:] == 0).sum() == 0: - if fill_value is None: + pos, neg = nw.col("__pos__"), nw.col("__neg__") + if fill_value is None: + has_zero = bool(stats.select(((pos == 0) | (neg == 0)).any()).item()) + if has_zero is True: raise ValueError( "The proportion of one of the classes for a category in " "variable {} is zero, and log of zero is not defined".format( variable ) ) - else: - pos[pos[:] == 0] = fill_value - neg[neg[:] == 0] = fill_value + else: + pos = nw.when(pos == 0).then(fill_value).otherwise(pos) + neg = nw.when(neg == 0).then(fill_value).otherwise(neg) - woe = np.log(pos / neg) - return pos, neg, woe + return stats.select( + "__category__", + pos.alias("__pos__"), + neg.alias("__neg__"), + (pos / neg).log().alias("__woe__"), + ) @Substitution( @@ -270,50 +289,19 @@ def fit(self, X: IntoDataFrame, y: IntoSeries): encoder_dict_ = {} vars_that_fail = [] - # pandas uses _calculate_woe(); other backends count the negative class - # as total minus positive, so they need one group_by instead of two - if nwd.is_pandas_dataframe(X): - for var in variables_: - try: - _, _, woe = self._calculate_woe(X, y, var, self.fill_value) - encoder_dict_[var] = woe.to_dict() - except ValueError: - vars_that_fail.append(var) - else: - nw_Xy = add_target_to_X(nw.from_native(X, eager_only=True), y) - - total_pos = nw_Xy[TARGET_NAME].sum() - total_neg = len(nw_Xy) - total_pos - - for var in variables_: - grouped = ( - nw_Xy.group_by(var, drop_null_keys=True) - .agg( - nw.col(TARGET_NAME).sum().alias("__pos_n__"), - nw.len().alias("__n__"), - ) - .sort(var) - ) - categories = grouped.get_column(var).to_list() - pos = (grouped.get_column("__pos_n__") / total_pos).to_numpy() - neg = ( - (grouped.get_column("__n__") - grouped.get_column("__pos_n__")) - / total_neg - ).to_numpy() - - if (pos == 0).any() or (neg == 0).any(): - if self.fill_value is None: - vars_that_fail.append(var) - continue - pos = np.where(pos == 0, self.fill_value, pos) - neg = np.where(neg == 0, self.fill_value, neg) - - woe = np.log(pos / neg) - encoder_dict_[var] = dict(zip(categories, woe)) + for var in variables_: + try: + woe = self._calculate_woe(X, y, var, self.fill_value) + except ValueError: + vars_that_fail.append(var) + continue + encoder_dict_[var] = dict( + zip(woe["__category__"].to_list(), woe["__woe__"].to_list()) + ) if len(vars_that_fail) > 0: vars_that_fail_str = ( - ", ".join(vars_that_fail) + ", ".join(str(var) for var in vars_that_fail) if len(vars_that_fail) > 1 else vars_that_fail[0] ) diff --git a/feature_engine/selection/information_value.py b/feature_engine/selection/information_value.py index b58dbeb9d..58db37365 100644 --- a/feature_engine/selection/information_value.py +++ b/feature_engine/selection/information_value.py @@ -230,8 +230,12 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self.information_values_ = {} for var in self.variables_: - total_pos, total_neg, woe = self._calculate_woe(X, y, var) - iv = self._calculate_iv(total_pos, total_neg, woe) + woe = self._calculate_woe(X, y, var) + iv = self._calculate_iv( + woe["__pos__"].to_numpy(), + woe["__neg__"].to_numpy(), + woe["__woe__"].to_numpy(), + ) self.information_values_[var] = iv self.features_to_drop_ = [ diff --git a/tests/test_encoding/test_woe/test_woe_class.py b/tests/test_encoding/test_woe/test_woe_class.py index 9fcfa4169..55df43252 100644 --- a/tests/test_encoding/test_woe/test_woe_class.py +++ b/tests/test_encoding/test_woe/test_woe_class.py @@ -1,71 +1,60 @@ +import math import re -import numpy as np -import pandas as pd import pytest from feature_engine.encoding.woe import WoE +from tests.backend_helpers import frame_to_dict, make_series +DATA_ZERO = { + "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], +} -def test_woe_calculation(df_enc): - pos_exp = pd.Series({"A": 0.333333, "B": 0.333333, "C": 0.333333}) - neg_exp = pd.Series({"A": 0.285714, "B": 0.571429, "C": 0.142857}) - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe(df_enc, df_enc["target"], "var_A") +def test_woe_calculation(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) + woe = WoE()._calculate_woe(X, y, "var_A").to_native() - -def test_woe_error(): - df = { - "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], + # 6 positive and 14 negative cases + pos = [2 / 6, 2 / 6, 2 / 6] + neg = [4 / 14, 8 / 14, 2 / 14] + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), } - df = pd.DataFrame(df) - woe_class = WoE() + + +def test_woe_error(make_df): + X = make_df(DATA_ZERO) + y = make_series(make_df, DATA_ZERO["target"]) msg = ( "The proportion of one of the classes for a category in variable var_A " "is zero, and log of zero is not defined" ) with pytest.raises(ValueError, match=re.escape(msg)): - woe_class._calculate_woe(df, df["target"], "var_A") + WoE()._calculate_woe(X, y, "var_A") @pytest.mark.parametrize("fill_value", [1, 10, 0.1]) -def test_fill_value(fill_value): - df = { - "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], +def test_fill_value(make_df, fill_value): + X = make_df(DATA_ZERO) + y = make_series(make_df, DATA_ZERO["target"]) + + woe = WoE()._calculate_woe(X, y, "var_A", fill_value=fill_value).to_native() + + # 7 positive and 13 negative cases; C has no negatives and D no positives + pos = [2 / 7, 2 / 7, 3 / 7, fill_value] + neg = [7 / 13, 4 / 13, fill_value, 2 / 13] + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C", "D"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), } - df = pd.DataFrame(df) - - pos_exp = pd.Series( - { - "A": 0.2857142857142857, - "B": 0.2857142857142857, - "C": 0.42857142857142855, - "D": fill_value, - } - ) - neg_exp = pd.Series( - { - "A": 0.5384615384615384, - "B": 0.3076923076923077, - "C": fill_value, - "D": 0.15384615384615385, - } - ) - - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe( - df, df["target"], "var_A", fill_value=fill_value - ) - - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index 35a6cdb2a..3e90261d1 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -337,6 +337,16 @@ def test_on_numerical_variables(make_df, data_enc_numeric): } +def test_integer_column_names(data_enc): + # integer column names are pandas-only + X = pd.DataFrame({0: data_enc["var_A"], 1: data_enc["var_B"]}) + y = pd.Series(data_enc["target"]) + + encoder = WoEEncoder().fit(X, y) + + assert encoder.encoder_dict_ == {0: WOE_A, 1: WOE_B} + + def test_variables_cast_as_category(df_enc_category_dtypes): # pandas Categorical dtype has no direct polars equivalent. df = df_enc_category_dtypes.copy() From b5cb287fcaba9589b48d77067bfdf27081c2f932 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 15:00:58 +0200 Subject: [PATCH 7/7] Replace zero counts by 0.5 in WoE, remove fill_value, add variables_with_zero_counts_ Co-Authored-By: Claude Opus 5 --- docs/user_guide/encoding/WoEEncoder.rst | 49 +++++- feature_engine/encoding/woe.py | 86 +++------- feature_engine/selection/information_value.py | 2 +- .../test_encoding/test_woe/test_woe_class.py | 40 ++--- .../test_woe/test_woe_encoder.py | 148 +++++------------- 5 files changed, 125 insertions(+), 200 deletions(-) diff --git a/docs/user_guide/encoding/WoEEncoder.rst b/docs/user_guide/encoding/WoEEncoder.rst index 7b75e1a30..4f43cc576 100644 --- a/docs/user_guide/encoding/WoEEncoder.rst +++ b/docs/user_guide/encoding/WoEEncoder.rst @@ -112,8 +112,10 @@ This occurs when a category shows only 1 of the possible values of the target (e always takes 1 or 0). In practice, this happens mostly when a category has a low frequency in the dataset, that is, when only very few observations show that category. -To overcome this limitation, consider using a variable transformation method to group -those categories together, for example by using feature-engine's :class:`RareLabelEncoder()`. +A common way to obtain a WoE for these categories is to replace the zero count by 0.5, +which is what :class:`WoEEncoder()` does. Still, WoE values calculated from very few +observations are unreliable, so consider grouping infrequent categories first, for +example with feature-engine's :class:`RareLabelEncoder()`. Taking into account the above considerations, conducting a detailed exploratory data analysis (EDA) is essential as part of the data science and model-building process. @@ -162,9 +164,46 @@ with feature-engine's imputers. :class:`WoEEncoder()` will ignore unseen categories by default, in which case, they will be replaced by np.nan after the encoding. You have the option to make the encoder raise -an error instead, by setting `unseen='raise'`. You can also replace unseen categories -by an arbitrary value you need to define in `fill_value`, although we do not recommend -this option because it may lead to unpredictable results. +an error instead, by setting `unseen='raise'`. + +Categories with no positive or no negative cases +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +.. attention:: + + **New in version 2.0:** :class:`WoEEncoder()` used to raise an error when a category + had no positive or no negative cases, unless you set the parameter `fill_value`. + `fill_value` was removed. The encoder now replaces zero counts by 0.5 and lists the + affected variables in the attribute `variables_with_zero_counts_`. + +When a category has no positive or no negative cases in the training set, +:class:`WoEEncoder()` replaces the zero count by 0.5 to calculate the WoE, and stores the +names of the affected variables in `variables_with_zero_counts_`. In the following +example, the category red has only positive cases: + +.. code:: python + + import pandas as pd + from feature_engine.encoding import WoEEncoder + + X = pd.DataFrame( + {"colour": ["blue", "blue", "blue", "red", "red", "green", "green", "green"]} + ) + y = pd.Series([1, 0, 1, 1, 1, 0, 1, 0]) + + woe = WoEEncoder() + woe.fit(X, y) + + print(woe.encoder_dict_) + print(woe.variables_with_zero_counts_) + +There are 5 positive and 3 negative cases. Red has 2 positive cases and no negative +cases, so its WoE is log((2 / 5) / (0.5 / 3)) = 0.88: + +.. code:: python + + {'colour': {'blue': 0.1823215567939548, 'green': -1.203972804325936, 'red': 0.8754687373539001}} + ['colour'] Python example -------------- diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index 95b6c9589..8f435fa11 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -67,12 +67,12 @@ def _calculate_woe( X: IntoDataFrame, y: IntoSeries, variable: Union[str, int], - fill_value: Union[float, None] = None, ): """ Return a narwhals dataframe with one row per category of the variable and the columns __category__, __pos__ and __neg__, the fraction of positive and - negative cases, and __woe__, the weight of evidence. + negative cases, and __woe__, the weight of evidence. Also return whether any + category has no positive or no negative cases. """ # narwhals expressions need string column names, pandas allows integers col = nw.from_native(X, eager_only=True).get_column(variable) @@ -80,37 +80,26 @@ def _calculate_woe( total_pos = nw_Xy[TARGET_NAME].sum() total_neg = len(nw_Xy) - total_pos - stats = ( + counts = ( nw_Xy.group_by("__category__", drop_null_keys=True) .agg(nw.col(TARGET_NAME).sum().alias("__pos__"), nw.len().alias("__n__")) .sort("__category__") - .select( - "__category__", - (nw.col("__pos__") / total_pos).alias("__pos__"), - ((nw.col("__n__") - nw.col("__pos__")) / total_neg).alias("__neg__"), - ) + .with_columns((nw.col("__n__") - nw.col("__pos__")).alias("__neg__")) ) - pos, neg = nw.col("__pos__"), nw.col("__neg__") - if fill_value is None: - has_zero = bool(stats.select(((pos == 0) | (neg == 0)).any()).item()) - if has_zero is True: - raise ValueError( - "The proportion of one of the classes for a category in " - "variable {} is zero, and log of zero is not defined".format( - variable - ) - ) - else: - pos = nw.when(pos == 0).then(fill_value).otherwise(pos) - neg = nw.when(neg == 0).then(fill_value).otherwise(neg) - - return stats.select( + has_zero_counts = bool(counts.select(((pos == 0) | (neg == 0)).any()).item()) + + # the WoE is not defined for zero counts, so they are replaced by 0.5 + pos = nw.when(pos == 0).then(0.5).otherwise(pos) / total_pos + neg = nw.when(neg == 0).then(0.5).otherwise(neg) / total_neg + + woe = counts.select( "__category__", pos.alias("__pos__"), neg.alias("__neg__"), (pos / neg).log().alias("__woe__"), ) + return woe, has_zero_counts @Substitution( @@ -148,10 +137,10 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): **Note** - The log(0) is not defined and the division by 0 is not defined. Thus, if any of the - terms in the WoE equation are 0 for a given category, the encoder will return an - error. If this happens, try grouping less frequent categories. Alternatively, - you can now add a fill_value (see parameter below). + The WoE is not defined for categories with no positive or no negative cases. For + those categories, the encoder replaces the zero count by 0.5, and lists the + variables in `variables_with_zero_counts_`. Grouping infrequent categories before + the encoding reduces how often this happens. More details in the :ref:`User Guide `. @@ -165,17 +154,15 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): {unseen} - fill_value: int, float, default=None - When the numerator or denominator of the WoE calculation are zero, the WoE - calculation is not possible. If `fill_value` is None (recommended), an error - will be raised in those cases. Alternatively, fill_value will be used in place - of denominators or numerators that equal zero. - Attributes ---------- encoder_dict_: Dictionary with the WoE per variable. + variables_with_zero_counts_: + List of variables with categories that have no positive or no negative cases. + For those categories, 0.5 replaces the zero count to calculate the WoE. + {variables_} {feature_names_in_} @@ -257,17 +244,11 @@ def __init__( return_empty: bool = False, ignore_format: bool = False, unseen: str = "ignore", - fill_value: Union[int, float, None] = None, ) -> None: super().__init__(variables, return_empty, ignore_format) check_parameter_unseen(unseen, ["ignore", "raise"]) - if fill_value is not None and not isinstance(fill_value, (int, float)): - raise ValueError( - f"fill_value takes None, integer or float. Got {fill_value} instead." - ) self.unseen = unseen - self.fill_value = fill_value def fit(self, X: IntoDataFrame, y: IntoSeries): """ @@ -287,32 +268,18 @@ def fit(self, X: IntoDataFrame, y: IntoSeries): _check_contains_na(X, variables_) encoder_dict_ = {} - vars_that_fail = [] + variables_with_zero_counts_ = [] for var in variables_: - try: - woe = self._calculate_woe(X, y, var, self.fill_value) - except ValueError: - vars_that_fail.append(var) - continue + woe, has_zero_counts = self._calculate_woe(X, y, var) encoder_dict_[var] = dict( zip(woe["__category__"].to_list(), woe["__woe__"].to_list()) ) - - if len(vars_that_fail) > 0: - vars_that_fail_str = ( - ", ".join(str(var) for var in vars_that_fail) - if len(vars_that_fail) > 1 - else vars_that_fail[0] - ) - - raise ValueError( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - f"and hence the WoE can't be calculated: {vars_that_fail_str}." - ) + if has_zero_counts is True: + variables_with_zero_counts_.append(var) self.encoder_dict_ = encoder_dict_ + self.variables_with_zero_counts_ = variables_with_zero_counts_ self.variables_ = variables_ self._get_feature_names_in(X) return self @@ -340,9 +307,6 @@ def _more_tags(self): tags_dict = _return_tags() tags_dict["variables"] = "categorical" tags_dict["requires_y"] = True - # sklearn tests pass continuous arrays, which give zero denominators and - # make this transformer raise, so they are skipped - tags_dict["_skip_test"] = True return tags_dict def __sklearn_tags__(self): diff --git a/feature_engine/selection/information_value.py b/feature_engine/selection/information_value.py index 58db37365..fd0dd53fe 100644 --- a/feature_engine/selection/information_value.py +++ b/feature_engine/selection/information_value.py @@ -230,7 +230,7 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self.information_values_ = {} for var in self.variables_: - woe = self._calculate_woe(X, y, var) + woe, _ = self._calculate_woe(X, y, var) iv = self._calculate_iv( woe["__pos__"].to_numpy(), woe["__neg__"].to_numpy(), diff --git a/tests/test_encoding/test_woe/test_woe_class.py b/tests/test_encoding/test_woe/test_woe_class.py index 55df43252..3e6f16111 100644 --- a/tests/test_encoding/test_woe/test_woe_class.py +++ b/tests/test_encoding/test_woe/test_woe_class.py @@ -1,26 +1,22 @@ import math -import re import pytest from feature_engine.encoding.woe import WoE from tests.backend_helpers import frame_to_dict, make_series -DATA_ZERO = { - "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], -} - def test_woe_calculation(make_df, data_enc): X = make_df(data_enc) y = make_series(make_df, data_enc["target"]) - woe = WoE()._calculate_woe(X, y, "var_A").to_native() + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() # 6 positive and 14 negative cases pos = [2 / 6, 2 / 6, 2 / 6] neg = [4 / 14, 8 / 14, 2 / 14] + assert has_zero_counts is False assert isinstance(woe, make_df) assert frame_to_dict(woe) == { "__category__": ["A", "B", "C"], @@ -30,27 +26,21 @@ def test_woe_calculation(make_df, data_enc): } -def test_woe_error(make_df): - X = make_df(DATA_ZERO) - y = make_series(make_df, DATA_ZERO["target"]) - msg = ( - "The proportion of one of the classes for a category in variable var_A " - "is zero, and log of zero is not defined" - ) - with pytest.raises(ValueError, match=re.escape(msg)): - WoE()._calculate_woe(X, y, "var_A") - - -@pytest.mark.parametrize("fill_value", [1, 10, 0.1]) -def test_fill_value(make_df, fill_value): - X = make_df(DATA_ZERO) - y = make_series(make_df, DATA_ZERO["target"]) +def test_zero_counts_are_replaced_by_half(make_df): + data = { + "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], + } + X = make_df(data) + y = make_series(make_df, data["target"]) - woe = WoE()._calculate_woe(X, y, "var_A", fill_value=fill_value).to_native() + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() # 7 positive and 13 negative cases; C has no negatives and D no positives - pos = [2 / 7, 2 / 7, 3 / 7, fill_value] - neg = [7 / 13, 4 / 13, fill_value, 2 / 13] + pos = [2 / 7, 2 / 7, 3 / 7, 0.5 / 7] + neg = [7 / 13, 4 / 13, 0.5 / 13, 2 / 13] + assert has_zero_counts is True assert isinstance(woe, make_df) assert frame_to_dict(woe) == { "__category__": ["A", "B", "C", "D"], diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index 3e90261d1..95e148c38 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -28,22 +28,7 @@ ) -def _msg_zero_division(features): - return ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - f"and hence the WoE can't be calculated: {features}." - ) - - # init parameters -@pytest.mark.parametrize("fill_value", ["hola", [10], (1,)]) -def test_error_if_fill_value_not_allowed(fill_value): - msg = f"fill_value takes None, integer or float. Got {fill_value} instead." - with pytest.raises(ValueError, match=re.escape(msg)): - WoEEncoder(fill_value=fill_value) - - @pytest.mark.parametrize( "unseen", ["empanada", "encode", False, 1, None, ("raise", "ignore"), ["ignore"]] ) @@ -54,21 +39,13 @@ def test_error_if_unseen_not_permitted_value(unseen): @pytest.mark.parametrize( - "ignore_format, unseen, fill_value", - [ - (False, "ignore", None), - (True, "raise", 0.5), - (False, "raise", 10), - (True, "ignore", 0), - ], + "ignore_format, unseen", + [(False, "ignore"), (True, "raise"), (False, "raise"), (True, "ignore")], ) -def test_init_param_assignment(ignore_format, unseen, fill_value): - encoder = WoEEncoder( - ignore_format=ignore_format, unseen=unseen, fill_value=fill_value - ) +def test_init_param_assignment(ignore_format, unseen): + encoder = WoEEncoder(ignore_format=ignore_format, unseen=unseen) assert encoder.ignore_format is ignore_format assert encoder.unseen == unseen - assert encoder.fill_value == fill_value # fit and transform @@ -81,6 +58,7 @@ def test_automatically_select_variables(make_df, data_enc): Xt = encoder.transform(X) assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert encoder.variables_with_zero_counts_ == [] assert isinstance(Xt, make_df) assert frame_to_dict(Xt) == { "var_A": pytest.approx(VAR_A), @@ -192,103 +170,57 @@ def test_error_if_target_not_binary(make_df): encoder.fit(X, y) -def test_error_if_denominator_probability_is_zero_1_var(make_df): +def test_zero_counts_are_replaced_by_half(make_df): + # in var_A, C has no negative cases and D no positive cases data = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError, match=re.escape(_msg_zero_division("var_A"))): - encoder.fit( - make_df(data)[["var_A", "var_B"]], make_series(make_df, data["target"]) - ) - - data = { - "var_A": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_B": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError, match=re.escape(_msg_zero_division("var_B"))): - encoder.fit( - make_df(data)[["var_A", "var_B"]], make_series(make_df, data["target"]) - ) + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) + encoder = WoEEncoder().fit(X, y) + Xt = encoder.transform(X) -def test_error_if_denominator_probability_is_zero_2_vars(make_df): - data = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + # 7 positive and 13 negative cases + woe_a = { + "A": math.log((2 / 7) / (7 / 13)), + "B": math.log((2 / 7) / (4 / 13)), + "C": math.log((3 / 7) / (0.5 / 13)), + "D": math.log((0.5 / 7) / (2 / 13)), + } + woe_b = { + "A": math.log((2 / 7) / (8 / 13)), + "B": math.log((3 / 7) / (3 / 13)), + "C": math.log((2 / 7) / (2 / 13)), + } + assert encoder.encoder_dict_ == { + "var_A": pytest.approx(woe_a), + "var_B": pytest.approx(woe_b), + } + assert encoder.variables_with_zero_counts_ == ["var_A"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx([woe_a[v] for v in data["var_A"]]), + "var_B": pytest.approx([woe_b[v] for v in data["var_B"]]), } - encoder = WoEEncoder(variables=None) - msg = _msg_zero_division("var_A, var_C") - with pytest.raises(ValueError, match=re.escape(msg)): - encoder.fit(make_df(data), make_series(make_df, data["target"])) -def test_error_if_numerator_probability_is_zero(make_df): +def test_variables_with_zero_counts(make_df): + # category A of var_A and var_C has no negative cases data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - X = make_df(data) - y = make_series(make_df, data["target"]) - encoder = WoEEncoder(variables=None) - - msg = _msg_zero_division("var_A, var_C") - with pytest.raises(ValueError, match=re.escape(msg)): - encoder.fit(X, y) - - msg = _msg_zero_division("var_A") - with pytest.raises(ValueError, match=re.escape(msg)): - encoder.fit(X[["var_A", "var_B"]], y) - - -def test_fill_value(make_df): - data = { - "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], + "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - X = make_df(data) + X = make_df(data)[["var_A", "var_B", "var_C"]] y = make_series(make_df, data["target"]) - encoder = WoEEncoder(variables=None, fill_value=1) - encoder.fit(X, y) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -0.8472978603872037, - "D": 1.8718021769015913, - } - woe_exp_b = { - "A": -0.7672551527136673, - "B": 0.6190392084062234, - "C": 0.6190392084062234, - } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) + encoder = WoEEncoder().fit(X, y) - encoder = WoEEncoder(variables=None, fill_value=10) - encoder.fit(X, y) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -3.1498829533812494, - "D": 4.174387269895637, - } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) + assert encoder.variables_with_zero_counts_ == ["var_A", "var_C"] def test_error_if_contains_na_in_fit(make_df, data_enc_na):