diff --git a/docs/user_guide/encoding/WoEEncoder.rst b/docs/user_guide/encoding/WoEEncoder.rst index a25d83074..4f43cc576 100644 --- a/docs/user_guide/encoding/WoEEncoder.rst +++ b/docs/user_guide/encoding/WoEEncoder.rst @@ -112,8 +112,10 @@ This occurs when a category shows only 1 of the possible values of the target (e always takes 1 or 0). In practice, this happens mostly when a category has a low frequency in the dataset, that is, when only very few observations show that category. -To overcome this limitation, consider using a variable transformation method to group -those categories together, for example by using feature-engine's :class:`RareLabelEncoder()`. +A common way to obtain a WoE for these categories is to replace the zero count by 0.5, +which is what :class:`WoEEncoder()` does. Still, WoE values calculated from very few +observations are unreliable, so consider grouping infrequent categories first, for +example with feature-engine's :class:`RareLabelEncoder()`. Taking into account the above considerations, conducting a detailed exploratory data analysis (EDA) is essential as part of the data science and model-building process. @@ -162,9 +164,46 @@ with feature-engine's imputers. :class:`WoEEncoder()` will ignore unseen categories by default, in which case, they will be replaced by np.nan after the encoding. You have the option to make the encoder raise -an error instead, by setting `unseen='raise'`. You can also replace unseen categories -by an arbitrary value you need to define in `fill_value`, although we do not recommend -this option because it may lead to unpredictable results. +an error instead, by setting `unseen='raise'`. + +Categories with no positive or no negative cases +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +.. attention:: + + **New in version 2.0:** :class:`WoEEncoder()` used to raise an error when a category + had no positive or no negative cases, unless you set the parameter `fill_value`. + `fill_value` was removed. The encoder now replaces zero counts by 0.5 and lists the + affected variables in the attribute `variables_with_zero_counts_`. + +When a category has no positive or no negative cases in the training set, +:class:`WoEEncoder()` replaces the zero count by 0.5 to calculate the WoE, and stores the +names of the affected variables in `variables_with_zero_counts_`. In the following +example, the category red has only positive cases: + +.. code:: python + + import pandas as pd + from feature_engine.encoding import WoEEncoder + + X = pd.DataFrame( + {"colour": ["blue", "blue", "blue", "red", "red", "green", "green", "green"]} + ) + y = pd.Series([1, 0, 1, 1, 1, 0, 1, 0]) + + woe = WoEEncoder() + woe.fit(X, y) + + print(woe.encoder_dict_) + print(woe.variables_with_zero_counts_) + +There are 5 positive and 3 negative cases. Red has 2 positive cases and no negative +cases, so its WoE is log((2 / 5) / (0.5 / 3)) = 0.88: + +.. code:: python + + {'colour': {'blue': 0.1823215567939548, 'green': -1.203972804325936, 'red': 0.8754687373539001}} + ['colour'] Python example -------------- @@ -280,6 +319,41 @@ variable values: 686 -0.584173 female 22.000000 0 0 7.7250 -0.357528 0.012075 +With polars +~~~~~~~~~~~ + +:class:`WoEEncoder()` also works with polars dataframes: + +.. code:: python + + import polars as pl + from feature_engine.encoding import WoEEncoder + + X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + y = pl.Series([0,1,1,1,0]) + + woe = WoEEncoder() + woe.fit(X, y) + woe.transform(X) + +We see the resulting dataframe below: + +.. code:: text + + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ + + WoE in categorical and numerical variables ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ diff --git a/feature_engine/encoding/woe.py b/feature_engine/encoding/woe.py index bd1538a2c..8f435fa11 100644 --- a/feature_engine/encoding/woe.py +++ b/feature_engine/encoding/woe.py @@ -3,8 +3,8 @@ from typing import List, Union -import numpy as np -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, @@ -26,7 +26,11 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.dataframe_checks import _check_contains_na, check_X_y -from feature_engine.encoding._helper_functions import check_parameter_unseen +from feature_engine.encoding._helper_functions import ( + TARGET_NAME, + add_target_to_X, + check_parameter_unseen, +) from feature_engine.encoding.base_encoder import ( CategoricalInitMixin, CategoricalMethodsMixin, @@ -35,14 +39,16 @@ class WoE: - def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): + def _check_fit_input(self, X: IntoDataFrame, y: IntoSeries): """ Check that X is dataframe, and y a binary series with values 0 and 1. """ - X, y = check_X_y(X, y) + nw_X, y = check_X_y(X, y) + # with pandas, y takes the index of X + y_nw = add_target_to_X(nw_X, y)[TARGET_NAME] # check that y is binary - if y.nunique() != 2: + if y_nw.n_unique() != 2: raise ValueError( "This encoder is designed for binary classification. The target " "used has more than 2 unique values." @@ -50,38 +56,50 @@ def _check_fit_input(self, X: pd.DataFrame, y: pd.Series): # if target does not have values 0 and 1, we need to remap, to be able to # compute the averages. - if y.min() != 0 or y.max() != 1: - y = pd.Series(np.where(y == y.min(), 0, 1)) - return X, y + y_min, y_max = y_nw.min(), y_nw.max() + if y_min != 0 or y_max != 1: + y_nw = (y_nw != y_min).cast(nw.Int64()).alias("target") + + return X, y_nw.to_native() def _calculate_woe( self, - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y: IntoSeries, variable: Union[str, int], - fill_value: Union[float, None] = None, ): - total_pos = y.sum() - inverse_y = y.ne(1).copy() - total_neg = inverse_y.sum() - - pos = y.groupby(X[variable], observed=False).sum() / total_pos - neg = inverse_y.groupby(X[variable], observed=False).sum() / total_neg - - if not (pos[:] == 0).sum() == 0 or not (neg[:] == 0).sum() == 0: - if fill_value is None: - raise ValueError( - "The proportion of one of the classes for a category in " - "variable {} is zero, and log of zero is not defined".format( - variable - ) - ) - else: - pos[pos[:] == 0] = fill_value - neg[neg[:] == 0] = fill_value - - woe = np.log(pos / neg) - return pos, neg, woe + """ + Return a narwhals dataframe with one row per category of the variable and the + columns __category__, __pos__ and __neg__, the fraction of positive and + negative cases, and __woe__, the weight of evidence. Also return whether any + category has no positive or no negative cases. + """ + # narwhals expressions need string column names, pandas allows integers + col = nw.from_native(X, eager_only=True).get_column(variable) + nw_Xy = add_target_to_X(col.alias("__category__").to_frame(), y) + total_pos = nw_Xy[TARGET_NAME].sum() + total_neg = len(nw_Xy) - total_pos + + counts = ( + nw_Xy.group_by("__category__", drop_null_keys=True) + .agg(nw.col(TARGET_NAME).sum().alias("__pos__"), nw.len().alias("__n__")) + .sort("__category__") + .with_columns((nw.col("__n__") - nw.col("__pos__")).alias("__neg__")) + ) + pos, neg = nw.col("__pos__"), nw.col("__neg__") + has_zero_counts = bool(counts.select(((pos == 0) | (neg == 0)).any()).item()) + + # the WoE is not defined for zero counts, so they are replaced by 0.5 + pos = nw.when(pos == 0).then(0.5).otherwise(pos) / total_pos + neg = nw.when(neg == 0).then(0.5).otherwise(neg) / total_neg + + woe = counts.select( + "__category__", + pos.alias("__pos__"), + neg.alias("__neg__"), + (pos / neg).log().alias("__woe__"), + ) + return woe, has_zero_counts @Substitution( @@ -119,10 +137,10 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): **Note** - The log(0) is not defined and the division by 0 is not defined. Thus, if any of the - terms in the WoE equation are 0 for a given category, the encoder will return an - error. If this happens, try grouping less frequent categories. Alternatively, - you can now add a fill_value (see parameter below). + The WoE is not defined for categories with no positive or no negative cases. For + those categories, the encoder replaces the zero count by 0.5, and lists the + variables in `variables_with_zero_counts_`. Grouping infrequent categories before + the encoding reduces how often this happens. More details in the :ref:`User Guide `. @@ -136,17 +154,15 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): {unseen} - fill_value: int, float, default=None - When the numerator or denominator of the WoE calculation are zero, the WoE - calculation is not possible. If `fill_value` is None (recommended), an error - will be raised in those cases. Alternatively, fill_value will be used in place - of denominators or numerators that equal zero. - Attributes ---------- encoder_dict_: Dictionary with the WoE per variable. + variables_with_zero_counts_: + List of variables with categories that have no positive or no negative cases. + For those categories, 0.5 replaces the zero count to calculate the WoE. + {variables_} {feature_names_in_} @@ -198,6 +214,28 @@ class WoEEncoder(CategoricalMethodsMixin, CategoricalInitMixin, WoE): 2 3 0.287682 3 4 -0.405465 4 5 -0.405465 + + With polars + + >>> import polars as pl + >>> from feature_engine.encoding import WoEEncoder + >>> X = pl.DataFrame(dict(x1 = [1,2,3,4,5], x2 = ["b", "b", "b", "a", "a"])) + >>> y = pl.Series([0,1,1,1,0]) + >>> woe = WoEEncoder() + >>> woe.fit(X, y) + >>> woe.transform(X) + shape: (5, 2) + ┌─────┬───────────┐ + │ x1 ┆ x2 │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════════╡ + │ 1 ┆ 0.287682 │ + │ 2 ┆ 0.287682 │ + │ 3 ┆ 0.287682 │ + │ 4 ┆ -0.405465 │ + │ 5 ┆ -0.405465 │ + └─────┴───────────┘ """ def __init__( @@ -206,29 +244,23 @@ def __init__( return_empty: bool = False, ignore_format: bool = False, unseen: str = "ignore", - fill_value: Union[int, float, None] = None, ) -> None: super().__init__(variables, return_empty, ignore_format) check_parameter_unseen(unseen, ["ignore", "raise"]) - if fill_value is not None and not isinstance(fill_value, (int, float)): - raise ValueError( - f"fill_value takes None, integer or float. Got {fill_value} instead." - ) self.unseen = unseen - self.fill_value = fill_value - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Learn the WoE. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training input samples. Can be the entire dataframe, not just the categorical variables. - y: pandas series. + y: Series. Target, must be binary. """ X, y = self._check_fit_input(X, y) @@ -236,63 +268,45 @@ def fit(self, X: pd.DataFrame, y: pd.Series): _check_contains_na(X, variables_) encoder_dict_ = {} - vars_that_fail = [] + variables_with_zero_counts_ = [] for var in variables_: - try: - _, _, woe = self._calculate_woe(X, y, var, self.fill_value) - encoder_dict_[var] = woe.to_dict() - except ValueError: - vars_that_fail.append(var) - - if len(vars_that_fail) > 0: - vars_that_fail_str = ( - ", ".join(vars_that_fail) - if len(vars_that_fail) > 1 - else vars_that_fail[0] - ) - - raise ValueError( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - f"and hence the WoE can't be calculated: {vars_that_fail_str}." + woe, has_zero_counts = self._calculate_woe(X, y, var) + encoder_dict_[var] = dict( + zip(woe["__category__"].to_list(), woe["__woe__"].to_list()) ) + if has_zero_counts is True: + variables_with_zero_counts_.append(var) self.encoder_dict_ = encoder_dict_ + self.variables_with_zero_counts_ = variables_with_zero_counts_ self.variables_ = variables_ self._get_feature_names_in(X) return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Replace categories with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The dataset to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features]. + X_new: dataframe of shape = [n_samples, n_features]. The dataframe containing the categories replaced by numbers. """ - X = self._check_transform_input_and_state(X) + nw_X = self._check_transform_input_and_state(X) _check_contains_na(X, self.variables_) - X = self._encode(X) + X = self._encode(nw_X) return X def _more_tags(self): tags_dict = _return_tags() tags_dict["variables"] = "categorical" tags_dict["requires_y"] = True - # in the current format, the tests are performed using continuous np.arrays - # this means that when we encode some of the values, the denominator is 0 - # and this the transformer raises an error, and the test fails. - # For this reason, most sklearn tests will fail. And it has nothing to - # do with the class not being compatible, it is just that the inputs passed - # are not suitable - tags_dict["_skip_test"] = True return tags_dict def __sklearn_tags__(self): diff --git a/feature_engine/selection/information_value.py b/feature_engine/selection/information_value.py index b58dbeb9d..fd0dd53fe 100644 --- a/feature_engine/selection/information_value.py +++ b/feature_engine/selection/information_value.py @@ -230,8 +230,12 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self.information_values_ = {} for var in self.variables_: - total_pos, total_neg, woe = self._calculate_woe(X, y, var) - iv = self._calculate_iv(total_pos, total_neg, woe) + woe, _ = self._calculate_woe(X, y, var) + iv = self._calculate_iv( + woe["__pos__"].to_numpy(), + woe["__neg__"].to_numpy(), + woe["__woe__"].to_numpy(), + ) self.information_values_[var] = iv self.features_to_drop_ = [ diff --git a/tests/test_encoding/test_woe/test_woe_class.py b/tests/test_encoding/test_woe/test_woe_class.py index f26253786..3e6f16111 100644 --- a/tests/test_encoding/test_woe/test_woe_class.py +++ b/tests/test_encoding/test_woe/test_woe_class.py @@ -1,66 +1,50 @@ -import numpy as np -import pandas as pd +import math + import pytest from feature_engine.encoding.woe import WoE - - -def test_woe_calculation(df_enc): - pos_exp = pd.Series({"A": 0.333333, "B": 0.333333, "C": 0.333333}) - neg_exp = pd.Series({"A": 0.285714, "B": 0.571429, "C": 0.142857}) - - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe(df_enc, df_enc["target"], "var_A") - - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) - - -def test_woe_error(): - df = { - "var_A": ["B"] * 9 + ["A"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], +from tests.backend_helpers import frame_to_dict, make_series + + +def test_woe_calculation(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) + + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() + + # 6 positive and 14 negative cases + pos = [2 / 6, 2 / 6, 2 / 6] + neg = [4 / 14, 8 / 14, 2 / 14] + assert has_zero_counts is False + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), } - df = pd.DataFrame(df) - woe_class = WoE() - - with pytest.raises(ValueError): - woe_class._calculate_woe(df, df["target"], "var_A") -@pytest.mark.parametrize("fill_value", [1, 10, 0.1]) -def test_fill_value(fill_value): - df = { +def test_zero_counts_are_replaced_by_half(make_df): + data = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = pd.DataFrame(df) - - pos_exp = pd.Series( - { - "A": 0.2857142857142857, - "B": 0.2857142857142857, - "C": 0.42857142857142855, - "D": fill_value, - } - ) - neg_exp = pd.Series( - { - "A": 0.5384615384615384, - "B": 0.3076923076923077, - "C": fill_value, - "D": 0.15384615384615385, - } - ) - - woe_class = WoE() - pos, neg, woe = woe_class._calculate_woe( - df, df["target"], "var_A", fill_value=fill_value - ) - - pd.testing.assert_series_equal(pos, pos_exp, check_names=False) - pd.testing.assert_series_equal(neg, neg_exp, check_names=False) - pd.testing.assert_series_equal(np.log(pos_exp / neg_exp), woe, check_names=False) + X = make_df(data) + y = make_series(make_df, data["target"]) + + woe, has_zero_counts = WoE()._calculate_woe(X, y, "var_A") + woe = woe.to_native() + + # 7 positive and 13 negative cases; C has no negatives and D no positives + pos = [2 / 7, 2 / 7, 3 / 7, 0.5 / 7] + neg = [7 / 13, 4 / 13, 0.5 / 13, 2 / 13] + assert has_zero_counts is True + assert isinstance(woe, make_df) + assert frame_to_dict(woe) == { + "__category__": ["A", "B", "C", "D"], + "__pos__": pytest.approx(pos), + "__neg__": pytest.approx(neg), + "__woe__": pytest.approx([math.log(p / n) for p, n in zip(pos, neg)]), + } diff --git a/tests/test_encoding/test_woe/test_woe_encoder.py b/tests/test_encoding/test_woe/test_woe_encoder.py index a38caa6fa..95e148c38 100644 --- a/tests/test_encoding/test_woe/test_woe_encoder.py +++ b/tests/test_encoding/test_woe/test_woe_encoder.py @@ -1,4 +1,5 @@ import math +import re import numpy as np import pandas as pd @@ -6,102 +7,98 @@ from sklearn.exceptions import NotFittedError from feature_engine.encoding import WoEEncoder +from tests.backend_helpers import make_series, frame_to_dict + +WOE_A = { + "A": 0.15415067982725836, + "B": -0.5389965007326869, + "C": 0.8472978603872037, +} +WOE_B = { + "A": -0.5389965007326869, + "B": 0.15415067982725836, + "C": 0.8472978603872037, +} +VAR_A = [WOE_A["A"]] * 6 + [WOE_A["B"]] * 10 + [WOE_A["C"]] * 4 +VAR_B = [WOE_B["A"]] * 10 + [WOE_B["B"]] * 6 + [WOE_B["C"]] * 4 + +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) -VAR_A = [ - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] -VAR_B = [ - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - -0.5389965007326869, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.15415067982725836, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, - 0.8472978603872037, -] +# init parameters +@pytest.mark.parametrize( + "unseen", ["empanada", "encode", False, 1, None, ("raise", "ignore"), ["ignore"]] +) +def test_error_if_unseen_not_permitted_value(unseen): + msg = f"Parameter `unseen` takes only values ignore, raise. Got {unseen} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + WoEEncoder(unseen=unseen) -def test_automatically_select_variables(df_enc): - encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - X = encoder.transform(df_enc[["var_A", "var_B"]]) +@pytest.mark.parametrize( + "ignore_format, unseen", + [(False, "ignore"), (True, "raise"), (False, "raise"), (True, "ignore")], +) +def test_init_param_assignment(ignore_format, unseen): + encoder = WoEEncoder(ignore_format=ignore_format, unseen=unseen) + assert encoder.ignore_format is ignore_format + assert encoder.unseen == unseen - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, +# fit and transform +def test_automatically_select_variables(make_df, data_enc): + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert encoder.variables_with_zero_counts_ == [] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), } - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) -def test_user_passes_variables(df_enc): - encoder = WoEEncoder(variables=["var_A", "var_B"]) - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_or_array(make_df, data_enc, to_target): + # a list or numpy array target takes a different code path than a Series + X = make_df(data_enc)[["var_A", "var_B"]] + y = to_target(data_enc["target"]) - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B + encoder = WoEEncoder(variables=None) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + +def test_user_passes_variables(make_df, data_enc): + X = make_df(data_enc) + y = make_series(make_df, data_enc["target"]) + + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": data_enc["target"], } - pd.testing.assert_frame_equal(X, transf_df) _targets = [ @@ -112,279 +109,183 @@ def test_user_passes_variables(df_enc): @pytest.mark.parametrize("target", _targets) -def test_when_target_class_not_0_1(df_enc, target): - encoder = WoEEncoder(variables=["var_A", "var_B"]) - df_enc["target"] = target - encoder.fit(df_enc, df_enc["target"]) - X = encoder.transform(df_enc) - - # transformed dataframe - transf_df = df_enc.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B +def test_when_target_class_not_0_1(make_df, data_enc, target): + data = dict(data_enc) + data["target"] = target + X = make_df(data) + y = make_series(make_df, target) - assert encoder.encoder_dict_ == { - "var_A": { - "A": 0.15415067982725836, - "B": -0.5389965007326869, - "C": 0.8472978603872037, - }, - "var_B": { - "A": -0.5389965007326869, - "B": 0.15415067982725836, - "C": 0.8472978603872037, - }, + encoder = WoEEncoder(variables=["var_A", "var_B"]) + encoder.fit(X, y) + Xt = encoder.transform(X) + + assert encoder.encoder_dict_ == {"var_A": WOE_A, "var_B": WOE_B} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + "target": target, } - pd.testing.assert_frame_equal(X, transf_df) -def test_warn_if_transform_df_contains_categories_not_seen_in_fit(df_enc, df_enc_rare): +def test_warn_if_transform_df_contains_categories_not_seen_in_fit( + make_df, data_enc, data_enc_rare +): # test case 3: when dataset to be transformed contains categories not present # in training dataset + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_rare = make_df(data_enc_rare)[["var_A", "var_B"]] msg = "During the encoding, NaN values were introduced in the feature(s) var_A." - # check for error when rare_labels equals 'raise' - with pytest.warns(UserWarning) as record: - encoder = WoEEncoder(unseen="ignore") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) - - # check that at least one warning was raised (Pandas 3 may emit additional - # deprecation warnings) - assert len(record) >= 1 - # check that the message matches - assert any(r.message.args[0] == msg for r in record) - - # check for error when rare_labels equals 'raise' - with pytest.raises(ValueError) as record: - encoder = WoEEncoder(unseen="raise") - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - encoder.transform(df_enc_rare[["var_A", "var_B"]]) + # check for warning when unseen equals 'ignore' + encoder = WoEEncoder(unseen="ignore") + encoder.fit(X, y) + with pytest.warns(UserWarning, match=re.escape(msg)): + encoder.transform(X_rare) - # check that the error message matches - assert str(record.value) == msg + # check for error when unseen equals 'raise' + encoder = WoEEncoder(unseen="raise") + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.transform(X_rare) -def test_error_if_target_not_binary(): +def test_error_if_target_not_binary(make_df): # test case 4: the target is not binary - encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - -def test_error_if_denominator_probability_is_zero_1_var(): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." - ) - assert str(record.value) == msg - - df = { - "var_A": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_B": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_B." - ) - assert str(record.value) == msg - - -def test_error_if_denominator_probability_is_zero_2_vars(): - df = { + data = { "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + "target": [1, 1, 2, 2, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - assert str(record.value) == msg - - -def test_error_if_numerator_probability_is_zero(): - df = { - "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, - "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, - "target": [0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], - } - df = pd.DataFrame(df) encoder = WoEEncoder(variables=None) - - with pytest.raises(ValueError) as record: - encoder.fit(df, df["target"]) - - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A, var_C." - ) - assert str(record.value) == msg - - with pytest.raises(ValueError) as record: - encoder.fit(df[["var_A", "var_B"]], df["target"]) - msg = ( - "During the WoE calculation, some of the categories in the " - "following features contained 0 in the denominator or numerator, " - "and hence the WoE can't be calculated: var_A." + "This encoder is designed for binary classification. The target " + "used has more than 2 unique values." ) - assert str(record.value) == msg + with pytest.raises(ValueError, match=re.escape(msg)): + encoder.fit(X, y) -def test_fill_value(): - df = { +def test_zero_counts_are_replaced_by_half(make_df): + # in var_A, C has no negative cases and D no positive cases + data = { "var_A": ["A"] * 9 + ["B"] * 6 + ["C"] * 3 + ["D"] * 2, "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, "target": [1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 1, 1, 1, 0, 0], } - df = pd.DataFrame(df) - encoder = WoEEncoder(variables=None, fill_value=1) - encoder.fit(df, df["target"]) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -0.8472978603872037, - "D": 1.8718021769015913, + X = make_df(data)[["var_A", "var_B"]] + y = make_series(make_df, data["target"]) + + encoder = WoEEncoder().fit(X, y) + Xt = encoder.transform(X) + + # 7 positive and 13 negative cases + woe_a = { + "A": math.log((2 / 7) / (7 / 13)), + "B": math.log((2 / 7) / (4 / 13)), + "C": math.log((3 / 7) / (0.5 / 13)), + "D": math.log((0.5 / 7) / (2 / 13)), + } + woe_b = { + "A": math.log((2 / 7) / (8 / 13)), + "B": math.log((3 / 7) / (3 / 13)), + "C": math.log((2 / 7) / (2 / 13)), } - woe_exp_b = { - "A": -0.7672551527136673, - "B": 0.6190392084062234, - "C": 0.6190392084062234, + assert encoder.encoder_dict_ == { + "var_A": pytest.approx(woe_a), + "var_B": pytest.approx(woe_b), } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) - - encoder = WoEEncoder(variables=None, fill_value=10) - encoder.fit(df, df["target"]) - woe_exp_a = { - "A": -0.6337237600891445, - "B": -0.07410797215372196, - "C": -3.1498829533812494, - "D": 4.174387269895637, + assert encoder.variables_with_zero_counts_ == ["var_A"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx([woe_a[v] for v in data["var_A"]]), + "var_B": pytest.approx([woe_b[v] for v in data["var_B"]]), } - woe_exp = {"var_A": woe_exp_a, "var_B": woe_exp_b} - for var in ["var_A", "var_B"]: - for k, i in woe_exp[var].items(): - assert math.isclose(encoder.encoder_dict_[var][k], woe_exp[var][k]) -@pytest.mark.parametrize("fill_value", ["hola", [10]]) -def test_error_if_fill_value_not_allowed(fill_value): - with pytest.raises(ValueError): - WoEEncoder(fill_value=fill_value) +def test_variables_with_zero_counts(make_df): + # category A of var_A and var_C has no negative cases + data = { + "var_A": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "var_B": ["A"] * 10 + ["B"] * 6 + ["C"] * 4, + "var_C": ["A"] * 6 + ["B"] * 10 + ["C"] * 4, + "target": [1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 1, 1, 0, 0], + } + X = make_df(data)[["var_A", "var_B", "var_C"]] + y = make_series(make_df, data["target"]) + encoder = WoEEncoder().fit(X, y) -@pytest.mark.parametrize("fill_value", [0, 1, 10, 0.5, 0.002, None]) -def test_assigns_fill_value_at_init(fill_value): - encoder = WoEEncoder(fill_value=fill_value) - assert encoder.fill_value == fill_value + assert encoder.variables_with_zero_counts_ == ["var_A", "var_C"] -def test_error_if_contains_na_in_fit(df_enc_na): +def test_error_if_contains_na_in_fit(make_df, data_enc_na): # test case 9: when dataset contains na, fit method + X = make_df(data_enc_na)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_na["target"]) + encoder = WoEEncoder(variables=None) - with pytest.raises(ValueError) as record: - encoder.fit(df_enc_na[["var_A", "var_B"]], df_enc_na["target"]) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.fit(X, y) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - assert str(record.value) == msg +def test_error_if_df_contains_na_in_transform(make_df, data_enc, data_enc_na): + # test case 10: when dataset contains na, transform method + X = make_df(data_enc)[["var_A", "var_B"]] + y = make_series(make_df, data_enc["target"]) + X_na = make_df(data_enc_na)[["var_A", "var_B"]] -def test_error_if_df_contains_na_in_transform(df_enc, df_enc_na): - # test case 10: when dataset contains na, transform method} encoder = WoEEncoder(variables=None) - encoder.fit(df_enc[["var_A", "var_B"]], df_enc["target"]) - with pytest.raises(ValueError) as record: - encoder.transform(df_enc_na[["var_A", "var_B"]]) - msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." - ) - assert str(record.value) == msg + encoder.fit(X, y) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + encoder.transform(X_na) -def test_on_numerical_variables(df_enc_numeric): +def test_on_numerical_variables(make_df, data_enc_numeric): # ignore_format=True - encoder = WoEEncoder(variables=None, ignore_format=True) - encoder.fit(df_enc_numeric[["var_A", "var_B"]], df_enc_numeric["target"]) - X = encoder.transform(df_enc_numeric[["var_A", "var_B"]]) + X = make_df(data_enc_numeric)[["var_A", "var_B"]] + y = make_series(make_df, data_enc_numeric["target"]) - # transformed dataframe - transf_df = df_enc_numeric.copy() - transf_df["var_A"] = VAR_A - transf_df["var_B"] = VAR_B + encoder = WoEEncoder(variables=None, ignore_format=True) + encoder.fit(X, y) + Xt = encoder.transform(X) - # init params - assert encoder.variables is None # fit params assert encoder.variables_ == ["var_A", "var_B"] assert encoder.encoder_dict_ == { - "var_A": { - 1: 0.15415067982725836, - 2: -0.5389965007326869, - 3: 0.8472978603872037, - }, - "var_B": { - 1: -0.5389965007326869, - 2: 0.15415067982725836, - 3: 0.8472978603872037, - }, + "var_A": {1: WOE_A["A"], 2: WOE_A["B"], 3: WOE_A["C"]}, + "var_B": {1: WOE_B["A"], 2: WOE_B["B"], 3: WOE_B["C"]}, } assert encoder.n_features_in_ == 2 # transform params - pd.testing.assert_frame_equal(X, transf_df[["var_A", "var_B"]]) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_A": pytest.approx(VAR_A), + "var_B": pytest.approx(VAR_B), + } + + +def test_integer_column_names(data_enc): + # integer column names are pandas-only + X = pd.DataFrame({0: data_enc["var_A"], 1: data_enc["var_B"]}) + y = pd.Series(data_enc["target"]) + + encoder = WoEEncoder().fit(X, y) + + assert encoder.encoder_dict_ == {0: WOE_A, 1: WOE_B} def test_variables_cast_as_category(df_enc_category_dtypes): + # pandas Categorical dtype has no direct polars equivalent. df = df_enc_category_dtypes.copy() encoder = WoEEncoder(variables=None) encoder.fit(df[["var_A", "var_B"]], df["target"]) X = encoder.transform(df[["var_A", "var_B"]]) - # transformed dataframe transf_df = df.copy() transf_df["var_A"] = VAR_A transf_df["var_B"] = VAR_B @@ -393,27 +294,24 @@ def test_variables_cast_as_category(df_enc_category_dtypes): assert X["var_A"].dtypes.name == "float64" -@pytest.mark.parametrize( - "errors", ["empanada", False, 1, ("raise", "ignore"), ["ignore"]] -) -def test_error_if_rare_labels_not_permitted_value(errors): - with pytest.raises(ValueError): - WoEEncoder(unseen=errors) - - -def test_inverse_transform_raises_non_fitted_error(): - df1 = pd.DataFrame({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) +def test_inverse_transform_raises_non_fitted_error(make_df): + df1 = make_df({"words": ["dog", "dog", "cat", "cat", "cat", "bird"]}) + y = make_series(make_df, [0, 1, 0, 1, 1, 0]) enc = WoEEncoder() + msg = ( + "This WoEEncoder instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): + with pytest.raises(NotFittedError, match=re.escape(msg)): enc.inverse_transform(df1) - df1.loc[len(df1) - 1] = np.nan + df1_na = make_df({"words": ["dog", "dog", "cat", "cat", "cat", None]}) - with pytest.raises(ValueError): - enc.fit(df1, pd.Series([0, 1, 0, 1, 1, 0])) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + enc.fit(df1_na, y) # Test when fit is not called prior to transform. - with pytest.raises(NotFittedError): - enc.inverse_transform(df1) + with pytest.raises(NotFittedError, match=re.escape(msg)): + enc.inverse_transform(df1_na)