From a79e5d044ef58f8978211611e4c1907d2162eba6 Mon Sep 17 00:00:00 2001 From: Ojas Sharma Date: Tue, 25 Aug 2026 05:27:24 -0400 Subject: [PATCH 1/4] Migrate scaling module (MeanNormalisationScaler) to narwhals, add polars support Redone from scratch off the current narwhals-migration HEAD rather than rebased forward from #979: that branch predates the dataframe_checks rewrite (#989), the variable_handling rewrite (#978), and the creation base rewrite (#990), so the delta had grown too large to carry forward safely for a module this small. fit() replaces the pandas .mean()/.max()/.min() reductions with a single narwhals+numpy path: wrap via nw.from_native, extract the variables as one batched array (nw_X.select(variables_).to_numpy()), reduce with numpy. transform()/inverse_transform() extract each variable as its own 1D array via get_column().to_numpy(), do the elementwise (x - mean) / range (or the inverse) in numpy, and write each back via nw.new_series(same_name, ...) + with_columns() -- same-named series replace the existing column in place, same as polars, rather than adding a new one the way RelativeFeatures/ MathFeatures do for their derived columns. Benchmarked narwhals-expression vs. narwhals+numpy for both fit and transform, at 100/10k/200k rows and 3/20 variables, both backends, before choosing: numpy wins by 2x-73x at small/medium scale on both pandas and polars, and even at 200k rows/polars where narwhals-expr pulls ahead it's only by ~2x, well inside the range this migration has been treating as "not worth a backend split" (CyclicalFeatures/ GeoDistanceFeatures used ~1.7x+ as the bar for splitting; nothing here gets close). One unified path, no pandas/polars branch, matching RelativeFeatures' precedent. return_empty=True guarded explicitly (mean_/range_ default to {} when variables_ is empty) -- narwhals' select([]) collapses row count too, so .to_numpy() on it would reduce over zero rows, not zero columns. Same fix CyclicalFeatures needed for the same reason. Docstring and user-guide numbers were wrong before this PR touched them, found while verifying rather than assumed: the docstring's five example values were literally the raw pre-normalization np.random.seed(42) draws, never the actual transform() output, and the user guide's inverse_transform table showed Age as a bare int (20, 21, ...) when both the pre-migration and post-migration code have always produced float64 there (multiplying by a float range always promotes the dtype, confirmed by running the pre-migration code directly). Fixed both, added a "With polars" section per AGENTS.md's doc-sync rule. Tests rewritten to the single-parametrized-over-both-backends convention (make_df=[pd.DataFrame, pl.DataFrame]) rather than kept pandas-only; all prior coverage preserved, including both class names (MeanNormalisationScaler and the deprecated MeanNormalizationScaler alias) and the deferred-attribute-assignment regression test. Verified: full test suite run twice, once against this branch and once against the unmodified narwhals-migration HEAD (via git stash) -- identical 68 pre-existing, unrelated failures in both runs (none in scaling; confirmed by diffing the two failure lists directly, not just comparing counts), 2273 -> 2287 passed (the +14 is exactly this file's new parametrized test count minus its old one). flake8 and mypy clean. --- .../scaling/MeanNormalisationScaler.rst | 55 ++++++- feature_engine/scaling/mean_normalization.py | 98 +++++++++--- tests/test_scaling/test_mean_normalization.py | 150 ++++++++++-------- 3 files changed, 210 insertions(+), 93 deletions(-) diff --git a/docs/user_guide/scaling/MeanNormalisationScaler.rst b/docs/user_guide/scaling/MeanNormalisationScaler.rst index 30952f06c..2255d8d2d 100644 --- a/docs/user_guide/scaling/MeanNormalisationScaler.rst +++ b/docs/user_guide/scaling/MeanNormalisationScaler.rst @@ -137,11 +137,56 @@ In the following data, we see the scaled variables returned to their original re .. code:: python - Name City Age Height Marks dob - 0 tom London 20 1.80 0.9 2020-02-24 00:00:00 - 1 nick Manchester 21 1.77 0.8 2020-02-24 00:01:00 - 2 krish Liverpool 19 1.90 0.7 2020-02-24 00:02:00 - 3 jack Bristol 18 2.00 0.6 2020-02-24 00:03:00 + Name City Age Height Marks dob + 0 tom London 20.0 1.80 0.9 2020-02-24 00:00:00 + 1 nick Manchester 21.0 1.77 0.8 2020-02-24 00:01:00 + 2 krish Liverpool 19.0 1.90 0.7 2020-02-24 00:02:00 + 3 jack Bristol 18.0 2.00 0.6 2020-02-24 00:03:00 + +Note that **Age** comes back as a float, not the original integer: multiplying and +adding floats (the range and mean) always produces a float in both pandas and +polars, so the inverse transformation cannot restore the original integer dtype. + +With polars +----------- + +:class:`MeanNormalisationScaler()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.scaling import MeanNormalisationScaler + + df = pl.DataFrame( + { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Height": [1.80, 1.77, 1.90, 2.00], + "Marks": [0.9, 0.8, 0.7, 0.6], + } + ) + + scaler = MeanNormalisationScaler(variables=["Age", "Marks", "Height"]) + scaler.fit(df) + + print(scaler.transform(df)) + +The resulting values match those found with pandas: + +.. code:: text + + shape: (4, 5) + ┌───────┬────────────┬───────────┬───────────┬───────────┐ + │ Name ┆ City ┆ Age ┆ Height ┆ Marks │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ str ┆ str ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════╪════════════╪═══════════╪═══════════╪═══════════╡ + │ tom ┆ London ┆ 0.166667 ┆ -0.293478 ┆ 0.5 │ + │ nick ┆ Manchester ┆ 0.5 ┆ -0.423913 ┆ 0.166667 │ + │ krish ┆ Liverpool ┆ -0.166667 ┆ 0.141304 ┆ -0.166667 │ + │ jack ┆ Bristol ┆ -0.5 ┆ 0.576087 ┆ -0.5 │ + └───────┴────────────┴───────────┴───────────┴───────────┘ Additional resources diff --git a/feature_engine/scaling/mean_normalization.py b/feature_engine/scaling/mean_normalization.py index 620865735..51b7f4739 100644 --- a/feature_engine/scaling/mean_normalization.py +++ b/feature_engine/scaling/mean_normalization.py @@ -4,7 +4,8 @@ import warnings from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer from feature_engine._check_init_parameters.check_init_input_params import ( @@ -96,12 +97,36 @@ class MeanNormalisationScaler(BaseNumericalTransformer): >>> mns.fit(X) >>> X = mns.transform(X) >>> X.head() - x - 0 0.496714 - 1 -0.138264 - 2 0.647689 - 3 1.523030 - 4 -0.234153 + x + 0 0.051125 + 1 -0.071456 + 2 0.093623 + 3 0.518122 + 4 -0.084093 + + With polars: + + >>> import numpy as np + >>> import polars as pl + >>> from feature_engine.scaling import MeanNormalisationScaler + >>> np.random.seed(42) + >>> X = pl.DataFrame(dict(x = np.random.lognormal(size = 100))) + >>> mns = MeanNormalisationScaler() + >>> mns.fit(X) + >>> X = mns.transform(X) + >>> X.head() + shape: (5, 1) + ┌───────────┐ + │ x │ + │ --- │ + │ f64 │ + ╞═══════════╡ + │ 0.051125 │ + │ -0.071456 │ + │ 0.093623 │ + │ 0.518122 │ + │ -0.084093 │ + └───────────┘ """ def __init__( @@ -115,25 +140,36 @@ def __init__( self.variables = _check_variables_input_value(variables) self.return_empty = return_empty - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Finds the mean and value range of each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The training input samples. Can be the entire dataframe, not just the variables to transform. - y: pandas Series, default=None + y: Series, default=None It is not needed in this transformer. You can pass y or None. """ # check input dataframe X, variables_ = self._fit_setup(X) - mean_ = X[variables_].mean().to_dict() - range_ = (X[variables_].max() - X[variables_].min()).to_dict() + if len(variables_) == 0: + # return_empty=True can leave variables_ empty; narwhals' select([]) + # collapses row count too, so .to_numpy() would reduce over 0 rows. + mean_: dict = {} + range_: dict = {} + else: + values = nw.from_native(X, eager_only=True).select(variables_).to_numpy() + mean_arr = values.mean(axis=0) + range_arr = values.max(axis=0) - values.min(axis=0) + # .tolist() converts numpy scalars to plain Python int/float, + # matching the dtype the old pandas .to_dict() used to return. + mean_ = dict(zip(variables_, mean_arr.tolist())) + range_ = dict(zip(variables_, range_arr.tolist())) # check for constant columns constant_columns = [col for col, value in range_.items() if value == 0] @@ -150,18 +186,18 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): return self - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Transform the variables using mean normalisation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe + X_new: dataframe The dataframe with the transformed variables. """ @@ -169,22 +205,31 @@ def transform(self, X: pd.DataFrame) -> pd.DataFrame: X = self._check_transform_input_and_state(X) # transformation - X[self.variables_] = (X[self.variables_] - self.mean_) / self.range_ + nw_X = nw.from_native(X, eager_only=True) + new_series = [ + nw.new_series( + var, + (nw_X.get_column(var).to_numpy() - self.mean_[var]) / self.range_[var], + backend=nw_X.implementation, + ) + for var in self.variables_ + ] + nw_X = nw_X.with_columns(*new_series) - return X + return nw_X.to_native() - def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: + def inverse_transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Convert the data back to the original representation. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_tr: pandas dataframe + X_tr: dataframe The dataframe with the transformed variables. """ @@ -192,9 +237,18 @@ def inverse_transform(self, X: pd.DataFrame) -> pd.DataFrame: X = self._check_transform_input_and_state(X) # inverse transform - X[self.variables_] = X[self.variables_] * self.range_ + self.mean_ + nw_X = nw.from_native(X, eager_only=True) + new_series = [ + nw.new_series( + var, + nw_X.get_column(var).to_numpy() * self.range_[var] + self.mean_[var], + backend=nw_X.implementation, + ) + for var in self.variables_ + ] + nw_X = nw_X.with_columns(*new_series) - return X + return nw_X.to_native() # TODO: remove in version 2.1.0 diff --git a/tests/test_scaling/test_mean_normalization.py b/tests/test_scaling/test_mean_normalization.py index 807a8a9fc..cf9ae7f4b 100644 --- a/tests/test_scaling/test_mean_normalization.py +++ b/tests/test_scaling/test_mean_normalization.py @@ -1,6 +1,9 @@ import re +import narwhals as nw +import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError @@ -16,6 +19,28 @@ "To silence this warning, use MeanNormalisationScaler instead." ) +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], +} + + +def _none_to_nan(values): + # Missing values print as None for polars, NaN for pandas float columns + # - both mean "missing" here, so normalize both sides before comparing. + return [np.nan if v is None else v for v in values] + + +def assert_df_equal(X, expected: dict, abs_tol: float = 1e-4) -> None: + result = nw.from_native(X, eager_only=True).to_dict(as_series=False) + assert list(result.keys()) == list(expected.keys()) + for col, values in expected.items(): + assert _none_to_nan(result[col]) == pytest.approx( + _none_to_nan(values), abs=abs_tol, nan_ok=True + ) + @pytest.fixture( params=[MeanNormalisationScaler, MeanNormalizationScaler], @@ -37,117 +62,107 @@ def test_mean_normalization_scaler_raises_future_warning(): MeanNormalizationScaler() -def test_transforming_int_vars(transformer_class): - # input test case - df = pd.DataFrame( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transforming_int_vars(make_df, transformer_class): + df = make_df( { "var1": [1.0, 2.0, 3.0], "var2": [4.0, 5.0, 3.0], "var3": [40.0, 20.0, 30.0], } ) - # expected output - expected_df = pd.DataFrame( - { - "var1": [-0.5, 0.0, 0.5], - "var2": [0, 0.5, -0.5], - "var3": [0.5, -0.5, 0.0], - } - ) + expected = { + "var1": [-0.5, 0.0, 0.5], + "var2": [0, 0.5, -0.5], + "var3": [0.5, -0.5, 0.0], + } transformer = make_transformer(transformer_class, variables=None) X = transformer.fit_transform(df) + assert_df_equal(X, expected) - pd.testing.assert_frame_equal(X, expected_df) - - # test inverse_transform Xit = transformer.inverse_transform(X) - - pd.testing.assert_frame_equal(Xit, df) + assert_df_equal( + Xit, + {"var1": [1.0, 2.0, 3.0], "var2": [4.0, 5.0, 3.0], "var3": [40.0, 20.0, 30.0]}, + ) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_mean_normalization_plus_automatically_find_variables( - df_vartypes, transformer_class + make_df, transformer_class ): - # test case 1: automatically select variables - transformer = make_transformer(transformer_class, variables=None) - X = transformer.fit_transform(df_vartypes) + df = make_df(DATA) - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [0.16666, 0.5, -0.16666, -0.5] - transf_df["Marks"] = [0.49999, 0.16666, -0.16666, -0.5] + transformer = make_transformer(transformer_class, variables=None) + X = transformer.fit_transform(df) - # test init params assert transformer.variables is None - # test fit attr assert transformer.variables_ == ["Age", "Marks"] - assert transformer.n_features_in_ == 5 - # test transform output - pd.testing.assert_frame_equal(X, transf_df, rtol=10e-3) + assert transformer.n_features_in_ == 4 - # test inverse_transform - Xit = transformer.inverse_transform(X) + expected = dict(DATA) + expected["Age"] = [0.16667, 0.5, -0.16667, -0.5] + expected["Marks"] = [0.5, 0.16667, -0.16667, -0.5] + assert_df_equal(X, expected) - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") - Xit["Marks"] = Xit["Marks"].round(1) + Xit = transformer.inverse_transform(X) + assert_df_equal(Xit, DATA) - # test - pd.testing.assert_frame_equal(Xit, df_vartypes, rtol=10e-3) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_mean_normalization_plus_user_passes_var_list(make_df, transformer_class): + df = make_df(DATA) -def test_mean_normalization_plus_user_passes_var_list(df_vartypes, transformer_class): - # test case 2: user passes variables transformer = make_transformer(transformer_class, variables="Age") - X = transformer.fit_transform(df_vartypes) - - # expected output - transf_df = df_vartypes.copy() - transf_df["Age"] = [0.16666, 0.5, -0.16666, -0.5] + X = transformer.fit_transform(df) - # test init params assert transformer.variables == "Age" - # test fit attr assert transformer.variables_ == ["Age"] - assert transformer.n_features_in_ == 5 - # test transform output - pd.testing.assert_frame_equal(X, transf_df, rtol=10e-3) + assert transformer.n_features_in_ == 4 - # test inverse_transform - Xit = transformer.inverse_transform(X) + expected = dict(DATA) + expected["Age"] = [0.16667, 0.5, -0.16667, -0.5] + assert_df_equal(X, expected) - # convert numbers to original format. - Xit["Age"] = Xit["Age"].round().astype("int64") + Xit = transformer.inverse_transform(X) + assert_df_equal(Xit, DATA) - # test - pd.testing.assert_frame_equal(Xit, df_vartypes, rtol=10e-3) +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_raises_error_if_na_in_df(make_df, transformer_class): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) -def test_fit_raises_error_if_na_in_df(df_na, transformer_class): - # test case 3: when dataset contains na, fit method transformer = make_transformer(transformer_class) with pytest.raises(ValueError): transformer.fit(df_na) -def test_transform_raises_error_if_na_in_df(df_vartypes, df_na, transformer_class): - # test case 4: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform_raises_error_if_na_in_df(make_df, transformer_class): + data_na = dict(DATA) + data_na["Age"] = [20, None, 19, 18] + df_na = make_df(data_na) + transformer = make_transformer(transformer_class) - transformer.fit(df_vartypes) + transformer.fit(make_df(DATA)) with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(df_na) -def test_non_fitted_error(df_vartypes, transformer_class): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df, transformer_class): + df = make_df(DATA) transformer = make_transformer(transformer_class) with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(df) -def test_constant_columns_error(transformer_class): - # input test case - df = pd.DataFrame( +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_constant_columns_error(make_df, transformer_class): + df = make_df( { "var1": [1.0, 2.0, 3.0], "var2": [4.0, 5.0, 3.0], @@ -163,7 +178,9 @@ def test_constant_columns_error(transformer_class): def test_raises_non_fitted_error_when_error_during_fit(transformer_class): # constant column: fails after mean_/range_ would have been computed, at # the "check for constant columns" step - real regression guard for the - # deferred trailing-underscore attribute assignment. + # deferred trailing-underscore attribute assignment. Pandas-only: this + # check's own helper (check_raises_non_fitted_error_when_fit_fails) + # builds a pandas frame internally. df = pd.DataFrame( { "var1": [1.0, 2.0, 3.0], @@ -176,6 +193,7 @@ def test_raises_non_fitted_error_when_error_during_fit(transformer_class): def test_check_return_empty(transformer_class): + # check_return_empty itself is pandas-only (builds pd.DataFrame internally). transformer = make_transformer(transformer_class) if transformer_class is MeanNormalizationScaler: with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): From b0cb31610d25419573a1e3ae516e017659b9c5b3 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Mon, 14 Sep 2026 22:46:48 +0200 Subject: [PATCH 2/4] Use shared backend test fixtures and helpers in MeanNormalisationScaler tests Replace the file-local assert_df_equal/_none_to_nan helpers and parametrize decorators with the shared test structure: make_df fixture, isinstance(X, make_df) plus to_dict() checks (pytest.approx for floats), and pytest.raises(match=re.escape(msg)). Co-Authored-By: Claude Opus 5 --- tests/test_scaling/test_mean_normalization.py | 133 ++++++++---------- 1 file changed, 60 insertions(+), 73 deletions(-) diff --git a/tests/test_scaling/test_mean_normalization.py b/tests/test_scaling/test_mean_normalization.py index cf9ae7f4b..f204b3127 100644 --- a/tests/test_scaling/test_mean_normalization.py +++ b/tests/test_scaling/test_mean_normalization.py @@ -1,13 +1,11 @@ import re -import narwhals as nw -import numpy as np import pandas as pd -import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.scaling import MeanNormalisationScaler, MeanNormalizationScaler +from tests.backend_helpers import to_dict from tests.estimator_checks.fit_functionality_checks import check_return_empty from tests.estimator_checks.non_fitted_error_checks import ( check_raises_non_fitted_error_when_fit_fails, @@ -19,6 +17,11 @@ "To silence this warning, use MeanNormalisationScaler instead." ) +MSG_NA = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." +) + DATA = { "Name": ["tom", "nick", "krish", "jack"], "City": ["London", "Manchester", "Liverpool", "Bristol"], @@ -27,21 +30,6 @@ } -def _none_to_nan(values): - # Missing values print as None for polars, NaN for pandas float columns - # - both mean "missing" here, so normalize both sides before comparing. - return [np.nan if v is None else v for v in values] - - -def assert_df_equal(X, expected: dict, abs_tol: float = 1e-4) -> None: - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - assert list(result.keys()) == list(expected.keys()) - for col, values in expected.items(): - assert _none_to_nan(result[col]) == pytest.approx( - _none_to_nan(values), abs=abs_tol, nan_ok=True - ) - - @pytest.fixture( params=[MeanNormalisationScaler, MeanNormalizationScaler], ids=["MeanNormalisationScaler", "MeanNormalizationScaler"], @@ -62,117 +50,116 @@ def test_mean_normalization_scaler_raises_future_warning(): MeanNormalizationScaler() -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_transforming_int_vars(make_df, transformer_class): - df = make_df( - { - "var1": [1.0, 2.0, 3.0], - "var2": [4.0, 5.0, 3.0], - "var3": [40.0, 20.0, 30.0], - } - ) - expected = { - "var1": [-0.5, 0.0, 0.5], - "var2": [0, 0.5, -0.5], - "var3": [0.5, -0.5, 0.0], + data = { + "var1": [1.0, 2.0, 3.0], + "var2": [4.0, 5.0, 3.0], + "var3": [40.0, 20.0, 30.0], } transformer = make_transformer(transformer_class, variables=None) - X = transformer.fit_transform(df) - assert_df_equal(X, expected) + X = transformer.fit_transform(make_df(data)) + assert isinstance(X, make_df) + assert to_dict(X) == { + "var1": pytest.approx([-0.5, 0.0, 0.5]), + "var2": pytest.approx([0, 0.5, -0.5]), + "var3": pytest.approx([0.5, -0.5, 0.0]), + } Xit = transformer.inverse_transform(X) - assert_df_equal( - Xit, - {"var1": [1.0, 2.0, 3.0], "var2": [4.0, 5.0, 3.0], "var3": [40.0, 20.0, 30.0]}, - ) + assert isinstance(Xit, make_df) + assert to_dict(Xit) == {col: pytest.approx(values) for col, values in data.items()} -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_mean_normalization_plus_automatically_find_variables( make_df, transformer_class ): - df = make_df(DATA) - transformer = make_transformer(transformer_class, variables=None) - X = transformer.fit_transform(df) + X = transformer.fit_transform(make_df(DATA)) assert transformer.variables is None assert transformer.variables_ == ["Age", "Marks"] assert transformer.n_features_in_ == 4 - expected = dict(DATA) - expected["Age"] = [0.16667, 0.5, -0.16667, -0.5] - expected["Marks"] = [0.5, 0.16667, -0.16667, -0.5] - assert_df_equal(X, expected) + assert isinstance(X, make_df) + assert to_dict(X) == { + "Name": DATA["Name"], + "City": DATA["City"], + "Age": pytest.approx([0.16667, 0.5, -0.16667, -0.5], abs=1e-4), + "Marks": pytest.approx([0.5, 0.16667, -0.16667, -0.5], abs=1e-4), + } Xit = transformer.inverse_transform(X) - assert_df_equal(Xit, DATA) + assert isinstance(Xit, make_df) + assert to_dict(Xit) == { + "Name": DATA["Name"], + "City": DATA["City"], + "Age": pytest.approx(DATA["Age"]), + "Marks": pytest.approx(DATA["Marks"]), + } -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_mean_normalization_plus_user_passes_var_list(make_df, transformer_class): - df = make_df(DATA) - transformer = make_transformer(transformer_class, variables="Age") - X = transformer.fit_transform(df) + X = transformer.fit_transform(make_df(DATA)) assert transformer.variables == "Age" assert transformer.variables_ == ["Age"] assert transformer.n_features_in_ == 4 - expected = dict(DATA) - expected["Age"] = [0.16667, 0.5, -0.16667, -0.5] - assert_df_equal(X, expected) + assert isinstance(X, make_df) + assert to_dict(X) == { + "Name": DATA["Name"], + "City": DATA["City"], + "Age": pytest.approx([0.16667, 0.5, -0.16667, -0.5], abs=1e-4), + "Marks": DATA["Marks"], + } Xit = transformer.inverse_transform(X) - assert_df_equal(Xit, DATA) + assert isinstance(Xit, make_df) + assert to_dict(Xit) == { + "Name": DATA["Name"], + "City": DATA["City"], + "Age": pytest.approx(DATA["Age"]), + "Marks": DATA["Marks"], + } -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_fit_raises_error_if_na_in_df(make_df, transformer_class): data_na = dict(DATA) data_na["Age"] = [20, None, 19, 18] - df_na = make_df(data_na) transformer = make_transformer(transformer_class) - with pytest.raises(ValueError): - transformer.fit(df_na) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.fit(make_df(data_na)) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_transform_raises_error_if_na_in_df(make_df, transformer_class): data_na = dict(DATA) data_na["Age"] = [20, None, 19, 18] - df_na = make_df(data_na) transformer = make_transformer(transformer_class) transformer.fit(make_df(DATA)) - with pytest.raises(ValueError): - transformer.transform(df_na) + with pytest.raises(ValueError, match=re.escape(MSG_NA)): + transformer.transform(make_df(data_na)) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_non_fitted_error(make_df, transformer_class): - df = make_df(DATA) transformer = make_transformer(transformer_class) with pytest.raises(NotFittedError): - transformer.transform(df) + transformer.transform(make_df(DATA)) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_constant_columns_error(make_df, transformer_class): - df = make_df( - { - "var1": [1.0, 2.0, 3.0], - "var2": [4.0, 5.0, 3.0], - "var3": [7.0, 7.0, 7.0], - } - ) + data = { + "var1": [1.0, 2.0, 3.0], + "var2": [4.0, 5.0, 3.0], + "var3": [7.0, 7.0, 7.0], + } transformer = make_transformer(transformer_class) with pytest.raises(ValueError, match=re.escape("Division by zero is not allowed")): - transformer.fit(df) + transformer.fit(make_df(data)) def test_raises_non_fitted_error_when_error_during_fit(transformer_class): From e8ac204d1f2a479542613cc78092b5017a349b35 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Tue, 15 Sep 2026 12:06:08 +0200 Subject: [PATCH 3/4] Use frame_to_dict after the shared helper rename in #1045 Co-Authored-By: Claude Opus 5 --- tests/test_scaling/test_mean_normalization.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/tests/test_scaling/test_mean_normalization.py b/tests/test_scaling/test_mean_normalization.py index f204b3127..1bf3be458 100644 --- a/tests/test_scaling/test_mean_normalization.py +++ b/tests/test_scaling/test_mean_normalization.py @@ -5,7 +5,7 @@ from sklearn.exceptions import NotFittedError from feature_engine.scaling import MeanNormalisationScaler, MeanNormalizationScaler -from tests.backend_helpers import to_dict +from tests.backend_helpers import frame_to_dict from tests.estimator_checks.fit_functionality_checks import check_return_empty from tests.estimator_checks.non_fitted_error_checks import ( check_raises_non_fitted_error_when_fit_fails, @@ -60,7 +60,7 @@ def test_transforming_int_vars(make_df, transformer_class): transformer = make_transformer(transformer_class, variables=None) X = transformer.fit_transform(make_df(data)) assert isinstance(X, make_df) - assert to_dict(X) == { + assert frame_to_dict(X) == { "var1": pytest.approx([-0.5, 0.0, 0.5]), "var2": pytest.approx([0, 0.5, -0.5]), "var3": pytest.approx([0.5, -0.5, 0.0]), @@ -68,7 +68,9 @@ def test_transforming_int_vars(make_df, transformer_class): Xit = transformer.inverse_transform(X) assert isinstance(Xit, make_df) - assert to_dict(Xit) == {col: pytest.approx(values) for col, values in data.items()} + assert frame_to_dict(Xit) == { + col: pytest.approx(values) for col, values in data.items() + } def test_mean_normalization_plus_automatically_find_variables( @@ -82,7 +84,7 @@ def test_mean_normalization_plus_automatically_find_variables( assert transformer.n_features_in_ == 4 assert isinstance(X, make_df) - assert to_dict(X) == { + assert frame_to_dict(X) == { "Name": DATA["Name"], "City": DATA["City"], "Age": pytest.approx([0.16667, 0.5, -0.16667, -0.5], abs=1e-4), @@ -91,7 +93,7 @@ def test_mean_normalization_plus_automatically_find_variables( Xit = transformer.inverse_transform(X) assert isinstance(Xit, make_df) - assert to_dict(Xit) == { + assert frame_to_dict(Xit) == { "Name": DATA["Name"], "City": DATA["City"], "Age": pytest.approx(DATA["Age"]), @@ -108,7 +110,7 @@ def test_mean_normalization_plus_user_passes_var_list(make_df, transformer_class assert transformer.n_features_in_ == 4 assert isinstance(X, make_df) - assert to_dict(X) == { + assert frame_to_dict(X) == { "Name": DATA["Name"], "City": DATA["City"], "Age": pytest.approx([0.16667, 0.5, -0.16667, -0.5], abs=1e-4), @@ -117,7 +119,7 @@ def test_mean_normalization_plus_user_passes_var_list(make_df, transformer_class Xit = transformer.inverse_transform(X) assert isinstance(Xit, make_df) - assert to_dict(Xit) == { + assert frame_to_dict(Xit) == { "Name": DATA["Name"], "City": DATA["City"], "Age": pytest.approx(DATA["Age"]), From be922e591393d876ea421dfffa012a466469cd69 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Fri, 18 Sep 2026 15:06:14 +0200 Subject: [PATCH 4/4] Match errors, drop init asserts and rename a test in MeanNormalisationScaler tests Co-Authored-By: Claude Opus 5 --- tests/test_scaling/test_mean_normalization.py | 25 +++++++++++-------- 1 file changed, 15 insertions(+), 10 deletions(-) diff --git a/tests/test_scaling/test_mean_normalization.py b/tests/test_scaling/test_mean_normalization.py index 1bf3be458..5a8049fc6 100644 --- a/tests/test_scaling/test_mean_normalization.py +++ b/tests/test_scaling/test_mean_normalization.py @@ -50,7 +50,9 @@ def test_mean_normalization_scaler_raises_future_warning(): MeanNormalizationScaler() -def test_transforming_int_vars(make_df, transformer_class): +def test_transform_and_inverse_transform_numerical_variables( + make_df, transformer_class +): data = { "var1": [1.0, 2.0, 3.0], "var2": [4.0, 5.0, 3.0], @@ -79,7 +81,6 @@ def test_mean_normalization_plus_automatically_find_variables( transformer = make_transformer(transformer_class, variables=None) X = transformer.fit_transform(make_df(DATA)) - assert transformer.variables is None assert transformer.variables_ == ["Age", "Marks"] assert transformer.n_features_in_ == 4 @@ -105,7 +106,6 @@ def test_mean_normalization_plus_user_passes_var_list(make_df, transformer_class transformer = make_transformer(transformer_class, variables="Age") X = transformer.fit_transform(make_df(DATA)) - assert transformer.variables == "Age" assert transformer.variables_ == ["Age"] assert transformer.n_features_in_ == 4 @@ -148,7 +148,11 @@ def test_transform_raises_error_if_na_in_df(make_df, transformer_class): def test_non_fitted_error(make_df, transformer_class): transformer = make_transformer(transformer_class) - with pytest.raises(NotFittedError): + msg = ( + f"This {transformer_class.__name__} instance is not fitted yet. Call 'fit' " + "with appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): transformer.transform(make_df(DATA)) @@ -160,16 +164,17 @@ def test_constant_columns_error(make_df, transformer_class): } transformer = make_transformer(transformer_class) - with pytest.raises(ValueError, match=re.escape("Division by zero is not allowed")): + msg = ( + "The following variable(s) are constant: ['var3']. " + "Division by zero is not allowed. Please remove constant columns." + ) + with pytest.raises(ValueError, match=re.escape(msg)): transformer.fit(make_df(data)) def test_raises_non_fitted_error_when_error_during_fit(transformer_class): - # constant column: fails after mean_/range_ would have been computed, at - # the "check for constant columns" step - real regression guard for the - # deferred trailing-underscore attribute assignment. Pandas-only: this - # check's own helper (check_raises_non_fitted_error_when_fit_fails) - # builds a pandas frame internally. + # fit fails on the constant column after computing mean_ and range_; the + # shared check builds pandas frames, so this test is pandas-only df = pd.DataFrame( { "var1": [1.0, 2.0, 3.0],