From fe9f2a3fb8129d87f865a089d5d39c2adb4795bc Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Wed, 26 Aug 2026 00:55:55 +0200 Subject: [PATCH 1/4] Migrate GeometricWidthDiscretiser.fit() to narwhals, add polars support fit()'s only pandas dependency was X[var].min()/.max() to compute the geometric progression's min/max anchors - everything downstream (the np.power/np.r_/np.sort bin-edge math) was already plain numpy and needed no changes. Replaced the pandas indexing with nw.from_native(X, eager_only=True).get_column(var).min()/.max(), which returns a numpy/python float scalar on both backends and feeds np.power identically either way. Benchmarked old pandas-native fit() vs the new narwhals-on-pandas and narwhals-on-polars paths at 10k/50k/100k rows x 1/2/10 columns (200 iterations each, min/max dominate cost either way since bin-edge math is O(bins) not O(n)): - narwhals-on-pandas: 1.0-1.3x of pandas-native at realistic sizes (50k-100k rows); the 1.8x seen only at the smallest 10k-row/1-col case is sub-millisecond fixed per-call overhead. Minimal loss - merged into a single narwhals path, no is_pandas split. - narwhals-on-polars: ~0.35-0.7x of pandas-native (i.e. 1.4-2.8x *faster*), consistent with the sibling BaseDiscretiser.transform() migration finding polars faster at every size tested. Verified: diffed new fit() bin edges against the old pandas implementation across edge cases (skewed/normal/negative-and-positive distributions, two-point range, and the min==max degenerate case) on both backends - numerically identical (exact equality, not just close). Cross-checked full fit_transform() (both return_object and return_boundaries combinations) between pandas and polars inputs - identical output values. Manually reran the GeometricWidthDiscretiser user guide's house_prices worked example (binner_dict_ and interval width numbers) against real output to confirm the docs still match current behaviour (the precision example there was already fixed in #986, prior to this branch) before adding a new "With polars" section with verified output. tests/test_discretisation/test_geometric_width_discretiser.py: the dataframe-touching tests are now parametrized over pd.DataFrame/pl.DataFrame per AGENTS.md, replacing the pandas-only df_normal_dist/df_na/df_vartypes fixtures with local dicts so the same input produces and asserts the same output on both backends (bin edges, transform values via narwhals-agnostic extraction, dtype checks, and NA-error cases). Init-only param-validation tests are unchanged since they never touch a dataframe. flake8 and mypy clean. Module imports with pandas blocked (loaded standalone, since sibling discretiser files in this package aren't migrated yet and still import pandas at their own module level). sphinx -W build clean (only the pre-existing unrelated linkcode_resolve warning). Full tests/test_discretisation suite: 114 passed, same 5 pre-existing failures as the unmodified base branch (test_check_estimator_discretisers.py - sklearn's check_estimator feeds raw numpy arrays, rejected by check_X()'s dataframe-only contract since the narwhals migration; unrelated to this change). Co-Authored-By: Claude Sonnet 5 --- .../GeometricWidthDiscretiser.rst | 60 +++++++++++++++++ .../discretisation/geometric_width.py | 11 ++-- .../test_geometric_width_discretiser.py | 65 ++++++++++++++----- 3 files changed, 114 insertions(+), 22 deletions(-) diff --git a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst index 74e150763..940746d9d 100644 --- a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst +++ b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst @@ -144,6 +144,66 @@ In the following output, we see the interval limits determined for each variable 2212.974, inf]} +With polars +----------- + +:class:`GeometricWidthDiscretiser()` works in the same way with a polars dataframe: + +.. code:: python + + import numpy as np + import polars as pl + from feature_engine.discretisation import GeometricWidthDiscretiser + + np.random.seed(42) + df = pl.DataFrame({"x": np.random.randint(1, 100, 100).astype(float)}) + + disc = GeometricWidthDiscretiser(bins=10) + Xt = disc.fit_transform(df) + + print(Xt["x"].value_counts().sort("x")) + +The resulting bin counts: + +.. code:: text + + shape: (9, 2) + ┌─────┬───────┐ + │ x ┆ count │ + │ --- ┆ --- │ + │ i64 ┆ u32 │ + ╞═════╪═══════╡ + │ 0 ┆ 6 │ + │ 1 ┆ 3 │ + │ 3 ┆ 3 │ + │ 4 ┆ 1 │ + │ 5 ┆ 5 │ + │ 6 ┆ 9 │ + │ 7 ┆ 8 │ + │ 8 ┆ 25 │ + │ 9 ┆ 40 │ + └─────┴───────┘ + +And the fitted bin edges, matching what we'd get fitting on the same values with pandas: + +.. code:: python + + disc.binner_dict_ + +.. code:: python + + {'x': [-inf, + 3.573433146226546, + 4.475691865644366, + 5.895335641248283, + 8.129050213617685, + 11.643650760992958, + 17.173639757979174, + 25.874707744105372, + 39.565256521047, + 61.106419756718246, + inf]} + Interval width ~~~~~~~~~~~~~~ diff --git a/feature_engine/discretisation/geometric_width.py b/feature_engine/discretisation/geometric_width.py index 709381c71..41aa9bd18 100644 --- a/feature_engine/discretisation/geometric_width.py +++ b/feature_engine/discretisation/geometric_width.py @@ -1,7 +1,8 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -159,14 +160,14 @@ def __init__( self.return_empty = return_empty self.bins = bins - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the boundaries of the geometric width intervals / bins for each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None @@ -177,10 +178,12 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): X, variables_ = self._fit_setup(X) # fit + nw_X = nw.from_native(X, eager_only=True) binner_dict_ = {} for var in variables_: - min_, max_ = X[var].min(), X[var].max() + col = nw_X.get_column(var) + min_, max_ = col.min(), col.max() increment = np.power(max_ - min_, 1.0 / self.bins) bins = np.r_[ -np.inf, min_ + np.power(increment, np.arange(1, self.bins)), np.inf diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index 6a4b56c2d..3e2e1c43a 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -1,11 +1,27 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import GeometricWidthDiscretiser +def _normal_dist_data(): + np.random.seed(0) + mu, sigma = 0, 0.1 # mean and standard deviation + return {"var": list(np.random.normal(mu, sigma, 100))} + + +def _get_column_values(X, column): + return nw.from_native(X, eager_only=True).get_column(column).to_list() + + +def _get_column_dtype(X, column): + return nw.from_native(X, eager_only=True).get_column(column).dtype + + # test init params @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_object_not_bool(param): @@ -43,14 +59,19 @@ def test_correct_param_assignment_at_init(params): assert t.bins == param2 -def test_fit_and_transform_methods(df_normal_dist): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_and_transform_methods(make_df): + data = _normal_dist_data() + df = make_df(data) + transformer = GeometricWidthDiscretiser( bins=10, variables=None, return_object=False ) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(df) # manual calculation - min_, max_ = df_normal_dist["var"].min(), df_normal_dist["var"].max() + arr = np.array(data["var"]) + min_, max_ = arr.min(), arr.max() increment = np.power(max_ - min_, 1.0 / 10) bins = np.r_[-np.inf, min_ + np.power(increment, np.arange(1, 10)), np.inf] bins = np.sort(bins) @@ -58,34 +79,42 @@ def test_fit_and_transform_methods(df_normal_dist): # fit params assert (transformer.binner_dict_["var"] == bins).all() - # transform params - assert ( - X["var"] == pd.cut(df_normal_dist["var"], bins=bins, precision=7).cat.codes - ).all() + # transform params - ground truth from pandas.cut on the same bins; values + # must match regardless of which backend the input dataframe uses. + expected = list(pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes) + assert _get_column_values(X, "var") == expected -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_find_variables_and_return_as_object(make_df): + df = make_df(_normal_dist_data()) transformer = GeometricWidthDiscretiser(bins=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" + X = transformer.fit_transform(df) + assert _get_column_dtype(X, "var") == nw.Object -def test_error_if_input_df_contains_na_in_fit(df_na): - # test case 3: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_input_df_contains_na_in_fit(make_df): + df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) transformer = GeometricWidthDiscretiser() with pytest.raises(ValueError): transformer.fit(df_na) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): - # test case 4: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_input_df_contains_na_in_transform(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) + df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) + transformer = GeometricWidthDiscretiser() - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(df_na) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) transformer = GeometricWidthDiscretiser() with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(df) From 5ffabe1a2d10179de611df7a97d6ff4fe62553dc Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Mon, 14 Sep 2026 22:43:08 +0200 Subject: [PATCH 2/4] Use shared backend test fixtures and helpers in GeometricWidthDiscretiser tests Replace the file-local _normal_dist_data/_get_column_values/_get_column_dtype helpers with the data_normal_dist fixture, make_df, isinstance(X, make_df) plus to_dict() checks, missing values written as None, and pytest.raises(match=re.escape(msg)). Co-Authored-By: Claude Opus 5 --- .../test_geometric_width_discretiser.py | 56 +++++++------------ 1 file changed, 21 insertions(+), 35 deletions(-) diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index 3e2e1c43a..30da6bc52 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -1,25 +1,18 @@ +import re + import narwhals as nw import numpy as np import pandas as pd -import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import GeometricWidthDiscretiser +from tests.backend_helpers import to_dict - -def _normal_dist_data(): - np.random.seed(0) - mu, sigma = 0, 0.1 # mean and standard deviation - return {"var": list(np.random.normal(mu, sigma, 100))} - - -def _get_column_values(X, column): - return nw.from_native(X, eager_only=True).get_column(column).to_list() - - -def _get_column_dtype(X, column): - return nw.from_native(X, eager_only=True).get_column(column).dtype +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) # test init params @@ -59,18 +52,14 @@ def test_correct_param_assignment_at_init(params): assert t.bins == param2 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_fit_and_transform_methods(make_df): - data = _normal_dist_data() - df = make_df(data) - +def test_fit_and_transform_methods(make_df, data_normal_dist): transformer = GeometricWidthDiscretiser( bins=10, variables=None, return_object=False ) - X = transformer.fit_transform(df) + X = transformer.fit_transform(make_df(data_normal_dist)) # manual calculation - arr = np.array(data["var"]) + arr = np.array(data_normal_dist["var"]) min_, max_ = arr.min(), arr.max() increment = np.power(max_ - min_, 1.0 / 10) bins = np.r_[-np.inf, min_ + np.power(increment, np.arange(1, 10)), np.inf] @@ -81,38 +70,35 @@ def test_fit_and_transform_methods(make_df): # transform params - ground truth from pandas.cut on the same bins; values # must match regardless of which backend the input dataframe uses. - expected = list(pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes) - assert _get_column_values(X, "var") == expected + expected = pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes.tolist() + assert isinstance(X, make_df) + assert to_dict(X)["var"] == expected -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_automatically_find_variables_and_return_as_object(make_df): - df = make_df(_normal_dist_data()) +def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): transformer = GeometricWidthDiscretiser(bins=10, variables=None, return_object=True) - X = transformer.fit_transform(df) - assert _get_column_dtype(X, "var") == nw.Object + X = transformer.fit_transform(make_df(data_normal_dist)) + assert isinstance(X, make_df) + assert nw.from_native(X, eager_only=True).schema["var"] == nw.Object -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_input_df_contains_na_in_fit(make_df): - df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) + df_na = make_df({"Age": [20.0, 21.0, None, 23.0]}) transformer = GeometricWidthDiscretiser() - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=re.escape(MSG_NA)): transformer.fit(df_na) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_error_if_input_df_contains_na_in_transform(make_df): df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) - df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) + df_na = make_df({"Age": [20.0, 21.0, None, 23.0]}) transformer = GeometricWidthDiscretiser() transformer.fit(df) - with pytest.raises(ValueError): + with pytest.raises(ValueError, match=re.escape(MSG_NA)): transformer.transform(df_na) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_non_fitted_error(make_df): df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) transformer = GeometricWidthDiscretiser() From 6f95be9678d6fdd7efb73cd78a1a0ca1352325f8 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 12:06:26 +0200 Subject: [PATCH 3/4] Use frame_to_dict after the shared helper rename in #1045 Co-Authored-By: Claude Opus 5 --- tests/test_discretisation/test_geometric_width_discretiser.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index 30da6bc52..a5427c183 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -7,7 +7,7 @@ from sklearn.exceptions import NotFittedError from feature_engine.discretisation import GeometricWidthDiscretiser -from tests.backend_helpers import to_dict +from tests.backend_helpers import frame_to_dict MSG_NA = ( "Some of the variables in the dataset contain NaN. Check and " @@ -72,7 +72,7 @@ def test_fit_and_transform_methods(make_df, data_normal_dist): # must match regardless of which backend the input dataframe uses. expected = pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes.tolist() assert isinstance(X, make_df) - assert to_dict(X)["var"] == expected + assert frame_to_dict(X)["var"] == expected def test_automatically_find_variables_and_return_as_object(make_df, data_normal_dist): From 0ee3d53b30f24e43277b8737af65949cb6140115 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Fri, 18 Sep 2026 13:24:15 +0200 Subject: [PATCH 4/4] Match errors and name init test in GeometricWidthDiscretiser tests Co-Authored-By: Claude Opus 5 --- .../test_geometric_width_discretiser.py | 25 +++++++++++++------ 1 file changed, 17 insertions(+), 8 deletions(-) diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index a5427c183..82a70add2 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -15,33 +15,37 @@ ) -# test init params +# init parameters @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_object_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_object must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(return_object=param) @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_boundaries_not_bool(param): - with pytest.raises(ValueError): + msg = f"return_boundaries must be True or False. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(return_boundaries=param) @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 0, -1]) def test_raises_error_when_precision_not_int(param): - with pytest.raises(ValueError): + msg = f"precision must be a positive integer. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(precision=param) -@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}]) +@pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, None]) def test_raises_error_when_bins_not_int(param): - with pytest.raises(ValueError): + msg = f"bins must be an integer. Got {param} instead." + with pytest.raises(ValueError, match=re.escape(msg)): GeometricWidthDiscretiser(bins=param) @pytest.mark.parametrize("params", [(False, 1), (True, 10)]) -def test_correct_param_assignment_at_init(params): +def test_init_param_assignment(params): param1, param2 = params t = GeometricWidthDiscretiser( return_object=param1, return_boundaries=param1, precision=param2, bins=param2 @@ -52,6 +56,7 @@ def test_correct_param_assignment_at_init(params): assert t.bins == param2 +# fit and transform def test_fit_and_transform_methods(make_df, data_normal_dist): transformer = GeometricWidthDiscretiser( bins=10, variables=None, return_object=False @@ -102,5 +107,9 @@ def test_error_if_input_df_contains_na_in_transform(make_df): def test_non_fitted_error(make_df): df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) transformer = GeometricWidthDiscretiser() - with pytest.raises(NotFittedError): + msg = ( + "This GeometricWidthDiscretiser instance is not fitted yet. Call 'fit' " + "with appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): transformer.transform(df)