diff --git a/docs/user_guide/imputation/RandomSampleImputer.rst b/docs/user_guide/imputation/RandomSampleImputer.rst index 3f6dcb9ad..70a59332c 100644 --- a/docs/user_guide/imputation/RandomSampleImputer.rst +++ b/docs/user_guide/imputation/RandomSampleImputer.rst @@ -28,12 +28,12 @@ missing data and `seed` is the number you entered in the `random_state`. If `seed = 'observation'`, then the random_state should be a variable name or a list of variable names. The seed will be calculated observation per -observation, either by adding or multiplying the values of the variables -indicated in the `random_state`. Then, a value will be extracted from the train set -using that seed and used to replace the NAN in that particular observation. This is the -equivalent of `pandas.sample(1, random_state=var1+var2)` if the `seeding_method` is -set to `add` or `pandas.sample(1, random_state=var1*var2)` if the `seeding_method` -is set to `multiply`. +observation from the values of the variables indicated in the `random_state`. +Then, a value will be extracted from the train set using that seed and used to +replace the NAN in that particular observation. + +Observations with the same values in the `random_state` variables receive the same +imputation, regardless of their position in the dataframe. For example, if the observation shows variables colour: np.nan, height: 152, weight:52, and we set the imputer as: @@ -43,20 +43,16 @@ and we set the imputer as: RandomSampleImputer( random_state=['height', 'weight'], seed='observation', - seeding_method='add', ) -the np.nan in the variable colour will be replaced using pandas sample as follows: - -.. code:: python - - observation.sample(1, random_state=int(152+52)) +the np.nan in the variable colour will be replaced with a value extracted from the train +set, using a seed derived from the values 152 and 52. Any other observation with +height 152 and weight 52 will receive the same value. .. note:: - Note, if the variables indicated in the `random_state` list are not numerical - the imputer will return an error. In addition, the variables indicated as seed - should not contain missing values themselves. + The variables indicated in the `random_state` must be numerical, otherwise the + imputer will return an error. Missing values in those variables are treated as 0. With polars ----------- @@ -79,7 +75,6 @@ With polars variables=["LotFrontage"], random_state=["MSSubClass", "YrSold"], seed="observation", - seeding_method="add", ) imputer.fit(X_train) imputer.transform(X_train) @@ -146,8 +141,8 @@ First, let's load the data and separate it into train and test: ) In this example, we sample values at random, observation per observation, using as seed -the value of the variable 'MSSubClass' plus the value of the variable 'YrSold'. Note -that the seed's value is different for each observation. +the values of the variables 'MSSubClass' and 'YrSold'. Observations with the same values +in these variables receive the same imputed values. The :class:`RandomSampleImputer()` will impute all variables in the data, as we left the default value of the parameter `variables` to `None`. @@ -158,7 +153,6 @@ default value of the parameter `variables` to `None`. imputer = RandomSampleImputer( random_state=['MSSubClass', 'YrSold'], seed='observation', - seeding_method='add' ) # fit the imputer diff --git a/feature_engine/imputation/arbitrary_imputer.py b/feature_engine/imputation/arbitrary_imputer.py index 49e0674b8..be130fba9 100644 --- a/feature_engine/imputation/arbitrary_imputer.py +++ b/feature_engine/imputation/arbitrary_imputer.py @@ -154,7 +154,10 @@ def __init__( if isinstance(arbitrary_number, int) or isinstance(arbitrary_number, float): self.arbitrary_number = arbitrary_number else: - raise ValueError("arbitrary_number must be numeric of type int or float") + raise ValueError( + "arbitrary_number must be numeric of type int or float. " + f"Got {arbitrary_number} instead." + ) _check_numerical_dict(imputer_dict) diff --git a/feature_engine/imputation/categorical.py b/feature_engine/imputation/categorical.py index a3ddfe172..94b9e0d79 100644 --- a/feature_engine/imputation/categorical.py +++ b/feature_engine/imputation/categorical.py @@ -148,13 +148,26 @@ def __init__( return_object: bool = False, ignore_format: bool = False, ) -> None: - if imputation_method not in ["missing", "frequent"]: + if not isinstance(imputation_method, str) or imputation_method not in [ + "missing", + "frequent", + ]: raise ValueError( - "imputation_method takes only values 'missing' or 'frequent'" + "imputation_method takes only values 'missing' or 'frequent'. " + f"Got {imputation_method} instead." ) if not isinstance(ignore_format, bool): - raise ValueError("ignore_format takes only booleans True and False") + raise ValueError( + "ignore_format takes only booleans True and False. " + f"Got {ignore_format} instead." + ) + + if not isinstance(return_object, bool): + raise ValueError( + "return_object takes only booleans True and False. " + f"Got {return_object} instead." + ) self.imputation_method = imputation_method self.fill_value = fill_value diff --git a/feature_engine/imputation/drop_missing_data.py b/feature_engine/imputation/drop_missing_data.py index 969676171..129e0df8f 100644 --- a/feature_engine/imputation/drop_missing_data.py +++ b/feature_engine/imputation/drop_missing_data.py @@ -256,15 +256,15 @@ def _select_rows(self, X: IntoDataFrame, keep: bool) -> IntoDataFrame: # dropna(subset=[]) keeps every row: there are no variables to # evaluate missingness on, so nothing can ever be "missing". if keep is True: - return X - if nwd.is_pandas_dataframe(X): - return X.iloc[:0] + return X.to_native() return X.head(0).to_native() # Benchmarked: a numpy-backed mask beats both pandas' own axis=1 # isnull()/notna().sum() (a known-slow reduction) and the narwhals - # path below, so pandas keeps this dedicated fast path. - if nwd.is_pandas_dataframe(X): + # path below, so pandas keeps this dedicated fast path. X is the + # narwhals frame returned by check_X, so branch on its implementation. + if X.implementation.is_pandas(): + X = X.to_native() if self.threshold is not None: non_null_count = X[self.variables_].notna().to_numpy().sum(axis=1) mask = non_null_count >= len(self.variables_) * self.threshold diff --git a/feature_engine/imputation/end_tail.py b/feature_engine/imputation/end_tail.py index 27ab98001..677a43b68 100644 --- a/feature_engine/imputation/end_tail.py +++ b/feature_engine/imputation/end_tail.py @@ -173,16 +173,23 @@ def __init__( return_empty: bool = False, ) -> None: - if imputation_method not in ["gaussian", "iqr", "max"]: + if not isinstance(imputation_method, str) or imputation_method not in [ + "gaussian", + "iqr", + "max", + ]: raise ValueError( - "imputation_method takes only values 'gaussian', 'iqr' or 'max'" + "imputation_method takes only values 'gaussian', 'iqr' or 'max'. " + f"Got {imputation_method} instead." ) - if tail not in ["right", "left"]: - raise ValueError("tail takes only values 'right' or 'left'") + if not isinstance(tail, str) or tail not in ["right", "left"]: + raise ValueError( + f"tail takes only values 'right' or 'left'. Got {tail} instead." + ) - if fold <= 0: - raise ValueError("fold takes only positive numbers") + if not isinstance(fold, (int, float)) or isinstance(fold, bool) or fold <= 0: + raise ValueError(f"fold takes only positive numbers. Got {fold} instead.") self.imputation_method = imputation_method self.tail = tail diff --git a/feature_engine/imputation/mean_median.py b/feature_engine/imputation/mean_median.py index 29e9a402e..a809ca665 100644 --- a/feature_engine/imputation/mean_median.py +++ b/feature_engine/imputation/mean_median.py @@ -136,8 +136,14 @@ def __init__( return_empty: bool = False, ) -> None: - if imputation_method not in ["median", "mean"]: - raise ValueError("imputation_method takes only values 'median' or 'mean'") + if not isinstance(imputation_method, str) or imputation_method not in [ + "median", + "mean", + ]: + raise ValueError( + "imputation_method takes only values 'median' or 'mean'. " + f"Got {imputation_method} instead." + ) self.imputation_method = imputation_method self.variables = _check_variables_input_value(variables) diff --git a/feature_engine/imputation/missing_indicator.py b/feature_engine/imputation/missing_indicator.py index 78d6b5e0f..45aa6f719 100644 --- a/feature_engine/imputation/missing_indicator.py +++ b/feature_engine/imputation/missing_indicator.py @@ -142,7 +142,10 @@ def __init__( ) -> None: if not isinstance(missing_only, bool): - raise ValueError("missing_only takes values True or False") + raise ValueError( + "missing_only takes values True or False. " + f"Got {missing_only} instead." + ) self.variables = _check_variables_input_value(variables) self.missing_only = missing_only diff --git a/feature_engine/imputation/random_sample.py b/feature_engine/imputation/random_sample.py index 62ed441ca..bc9804217 100644 --- a/feature_engine/imputation/random_sample.py +++ b/feature_engine/imputation/random_sample.py @@ -1,6 +1,7 @@ # Authors: Soledad Galli # License: BSD 3 clause +import hashlib from typing import List, Optional, Union import narwhals.dependencies as nwd @@ -32,21 +33,27 @@ from feature_engine.variable_handling import check_all_variables, find_all_variables -# for RandomSampleImputer -def _define_seed( - X: IntoDataFrame, - index: int, - seed_variables: Union[str, int, List[Union[str, int]]], - how: str = "add", -) -> int: - # Pandas-only: relies on .loc label-based row access, so it is only - # called from the pandas branch of transform(), where X is already - # confirmed to be a pandas dataframe. - if how == "add": - internal_seed = int(np.round(X.loc[index, seed_variables].sum(), 0)) - elif how == "multiply": - internal_seed = int(np.round(X.loc[index, seed_variables].product(), 0)) - return internal_seed +def _hash_seeds(values) -> np.ndarray: + """Return one seed per row, in [0, 2**32), derived from the row's values. + + Rows with the same values get the same seed, regardless of their position. + Values are compared as floats (25 and 25.0 are equal) and missing values + count as 0. hashlib, unlike hash(), gives the same seed in every session. + """ + values = np.asarray(values, dtype="float64") + values = values.reshape(len(values), -1) + # + 0.0 turns -0.0 into 0.0, so both give the same bytes + values = np.where(np.isnan(values), 0.0, values) + 0.0 + values = np.ascontiguousarray(values, dtype=" None: - if seed not in ["general", "observation"]: - raise ValueError("seed takes only values 'general' or 'observation'") - - if seeding_method not in ["add", "multiply"]: - raise ValueError("seeding_method takes only values 'add' or 'multiply'") + if not isinstance(seed, str) or seed not in ["general", "observation"]: + raise ValueError( + "seed takes only values 'general' or 'observation'. " + f"Got {seed} instead." + ) if seed == "general" and random_state: if not isinstance(random_state, int): raise ValueError( - "if seed == 'general' then random_state must take an integer" + "if seed == 'general' then random_state must take an integer. " + f"Got {random_state} instead." ) if seed == "observation" and not random_state: raise ValueError( "if seed == 'observation' the random state must take the name of one " - "or more variables which will be used to seed the imputer" + "or more variables which will be used to seed the imputer. " + f"Got {random_state} instead." ) self.variables = _check_variables_input_value(variables) @@ -200,7 +205,6 @@ def __init__( self.random_state = random_state self.seed = seed - self.seeding_method = seeding_method def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ @@ -244,7 +248,7 @@ def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): ): raise ValueError( "There are variables assigned as random state which are not part " - "of the training dataframe." + f"of the training dataframe. Got {self.random_state} instead." ) self.random_state = random_state @@ -308,26 +312,21 @@ def _transform_pandas(self, X): # random sampling observation per observation elif self.seed == "observation" and self.random_state: + # seeds come from the values before any variable is imputed; rows are + # addressed by position, so duplicated index labels don't matter + seeds = _hash_seeds(X[self.random_state].to_numpy()) for feature in self.variables_: - if X[feature].isnull().sum() > 0: - - # loop over each observation with missing data - for i in X[X[feature].isnull()].index: - # find the seed using additional variables - internal_seed = _define_seed( - X, i, self.random_state, how=self.seeding_method - ) - - # extract 1 value at random - random_sample = ( - self.X_[feature] - .dropna() - .sample(1, replace=True, random_state=internal_seed) - ) - random_sample = random_sample.values[0] - - # replace the missing data point - X.loc[i, feature] = random_sample + is_null = X[feature].isnull().to_numpy() + if is_null.any(): + pool = self.X_[feature].dropna() + positions = np.flatnonzero(is_null) + random_values = [ + pool.sample( + 1, replace=True, random_state=int(seeds[pos]) + ).iloc[0] + for pos in positions + ] + X.iloc[positions, X.columns.get_loc(feature)] = random_values return X def _transform_narwhals(self, X): @@ -351,15 +350,8 @@ def _transform_narwhals(self, X): X = X.with_columns(col.scatter(positions, random_sample)) elif self.seed == "observation" and self.random_state: - # Vectorized stand-in for pandas' .loc-based per-row seed lookup: - # narwhals dataframes are positional (no row labels), so the seed - # for every row is computed up-front with numpy instead of in a - # per-row .loc lookup. - seed_values = X.select(self.random_state).to_numpy() - if self.seeding_method == "add": - internal_seeds = np.round(seed_values.sum(axis=1), 0).astype(int) - else: - internal_seeds = np.round(seed_values.prod(axis=1), 0).astype(int) + # seeds come from the values before any variable is imputed + internal_seeds = _hash_seeds(X.select(self.random_state).to_numpy()) for feature in self.variables_: col = X[feature] diff --git a/tests/test_imputation/conftest.py b/tests/test_imputation/conftest.py new file mode 100644 index 000000000..906215f99 --- /dev/null +++ b/tests/test_imputation/conftest.py @@ -0,0 +1,49 @@ +"""Data shared by the imputer tests. + +Each fixture returns a fresh dict, so tests can build the dataframe on the +backend under test with ``make_df(data)``. Missing values are written as None, +not np.nan: polars treats np.nan as a real float value (not a null), so +mean/std/quantile would not skip it, unlike pandas. None becomes a null on +both backends. +""" + +import datetime +import pytest + + +@pytest.fixture +def data_na(): + return { + "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], + "City": [ + "London", + "Manchester", + None, + None, + "London", + "London", + "Bristol", + "Manchester", + ], + "Studies": [ + "Bachelor", + "Bachelor", + None, + None, + "Bachelor", + "PhD", + "None", + "Masters", + ], + "Age": [20, 21, 19, None, 23, 40, 41, 37], + "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], + } + + +@pytest.fixture +def data_na_dob(data_na): + # dob is never null: exercises a datetime variable that missing_only=True + # should exclude from variables_. + dob = [datetime.datetime(2020, 2, 24, 0, i) for i in range(8)] + # returns a new dict with every key of data_na plus dob + return data_na | {"dob": dob} diff --git a/tests/test_imputation/test_arbitrary_imputer.py b/tests/test_imputation/test_arbitrary_imputer.py index 9766204ec..8d26ec3e0 100644 --- a/tests/test_imputation/test_arbitrary_imputer.py +++ b/tests/test_imputation/test_arbitrary_imputer.py @@ -1,105 +1,114 @@ -import narwhals as nw -import pandas as pd -import polars as pl +import re + import pytest from feature_engine.imputation import ArbitraryImputer, ArbitraryNumberImputer - -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", +from tests.backend_helpers import frame_to_dict, null_count + + +# init parameters +@pytest.mark.parametrize("arbitrary_number", ["arbitrary", [1], None]) +def test_error_when_arbitrary_number_not_numeric(arbitrary_number): + msg = ( + "arbitrary_number must be numeric of type int or float. " + f"Got {arbitrary_number} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryImputer(arbitrary_number=arbitrary_number) + + +@pytest.mark.parametrize( + "imputer_dict", [{"Age": "arbitrary_number"}, {"Age": 1, "Marks": [2]}] +) +def test_error_when_imputer_dict_values_not_numeric(imputer_dict): + msg = ( + "All values in the dictionary must be integer or float. " + f"Got {imputer_dict} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + ArbitraryImputer(imputer_dict=imputer_dict) + + +@pytest.mark.parametrize("imputer_dict", ["Age", ["Age", 1], 1]) +def test_error_when_imputer_dict_not_dict(imputer_dict): + msg = ( + "The parameter can only take a dictionary or None. " + f"Got {imputer_dict} instead." + ) + with pytest.raises(TypeError, match=re.escape(msg)): + ArbitraryImputer(imputer_dict=imputer_dict) + + +@pytest.mark.parametrize( + "arbitrary_number, imputer_dict", + [ + (999, None), + (-1, None), + (0.5, {"Age": -42, "Marks": -999}), + (99, {"Age": 1.5}), ], - "Age": [20.0, 21.0, 19.0, None, 23.0, 40.0, 41.0, 37.0], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], -} - +) +def test_init_param_assignment(arbitrary_number, imputer_dict): + imputer = ArbitraryImputer( + arbitrary_number=arbitrary_number, imputer_dict=imputer_dict + ) + assert imputer.arbitrary_number == arbitrary_number + assert imputer.imputer_dict == imputer_dict -def _null_count(X, col) -> int: - return nw.from_native(X, eager_only=True)[col].is_null().sum() - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_impute_with_99_and_automatically_select_variables(make_df): - X = make_df(DATA) +# fit and transform +def test_impute_with_99_and_automatically_select_variables(make_df, data_na): imputer = ArbitraryImputer(arbitrary_number=99, variables=None) - X_transformed = imputer.fit_transform(X) - - # test init params - assert imputer.arbitrary_number == 99 - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age", "Marks"] - assert imputer.n_features_in_ == 4 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": 99, "Marks": 99} # selected variables should not contain NA, non-selected should still - assert _null_count(X_transformed, "Age") == 0 - assert _null_count(X_transformed, "Marks") == 0 - assert _null_count(X_transformed, "Name") > 0 - assert _null_count(X_transformed, "City") > 0 + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 - result = nw.from_native(X_transformed, eager_only=True).to_dict(as_series=False) - assert result["Age"] == [20.0, 21.0, 19.0, 99.0, 23.0, 40.0, 41.0, 37.0] - assert result["Marks"] == [0.9, 0.8, 0.7, 99.0, 0.3, 99.0, 0.8, 0.6] + result = frame_to_dict(X_transformed) + assert result["Age"] == [20, 21, 19, 99, 23, 40, 41, 37] + assert result["Marks"] == [0.9, 0.8, 0.7, 99, 0.3, 99, 0.8, 0.6] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_impute_with_1_and_single_variable_entered_by_user(make_df): - X = make_df(DATA) +def test_impute_with_1_and_single_variable_entered_by_user(make_df, data_na): imputer = ArbitraryImputer(arbitrary_number=-1, variables=["Age"]) - X_transformed = imputer.fit_transform(X) - - # test init params - assert imputer.arbitrary_number == -1 - assert imputer.variables == ["Age"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age"] - assert imputer.n_features_in_ == 4 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": -1} - assert _null_count(X_transformed, "Age") == 0 - result = nw.from_native(X_transformed, eager_only=True).to_dict(as_series=False) - assert result["Age"] == [20.0, 21.0, 19.0, -1.0, 23.0, 40.0, 41.0, 37.0] - - -def test_error_when_arbitrary_number_is_string(): - with pytest.raises(ValueError): - ArbitraryImputer(arbitrary_number="arbitrary") + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 19, -1, 23, 40, 41, 37] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_dictionary_of_imputation_values(make_df): - X = make_df(DATA) +def test_dictionary_of_imputation_values(make_df, data_na): imputer = ArbitraryImputer(imputer_dict={"Age": -42, "Marks": -999}) - X_transformed = imputer.fit_transform(X) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit params - assert imputer.n_features_in_ == 4 + assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": -42, "Marks": -999} - assert _null_count(X_transformed, "Age") == 0 - assert _null_count(X_transformed, "Marks") == 0 - assert _null_count(X_transformed, "Name") > 0 - assert _null_count(X_transformed, "City") > 0 - - result = nw.from_native(X_transformed, eager_only=True).to_dict(as_series=False) - assert result["Age"] == [20.0, 21.0, 19.0, -42.0, 23.0, 40.0, 41.0, 37.0] - assert result["Marks"] == [0.9, 0.8, 0.7, -999.0, 0.3, -999.0, 0.8, 0.6] - + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 -def test_imputer_error_when_dictionary_value_is_string(): - with pytest.raises(ValueError): - ArbitraryImputer(imputer_dict={"Age": "arbitrary_number"}) + result = frame_to_dict(X_transformed) + assert result["Age"] == [20, 21, 19, -42, 23, 40, 41, 37] + assert result["Marks"] == [0.9, 0.8, 0.7, -999, 0.3, -999, 0.8, 0.6] def test_arbitrary_number_imputer_is_deprecated(): @@ -107,4 +116,3 @@ def test_arbitrary_number_imputer_is_deprecated(): with pytest.warns(FutureWarning, match="ArbitraryNumberImputer was deprecated"): imputer = ArbitraryNumberImputer(arbitrary_number=99) assert isinstance(imputer, ArbitraryImputer) - assert imputer.arbitrary_number == 99 diff --git a/tests/test_imputation/test_categorical_imputer.py b/tests/test_imputation/test_categorical_imputer.py index 84539cfe0..94004bf24 100644 --- a/tests/test_imputation/test_categorical_imputer.py +++ b/tests/test_imputation/test_categorical_imputer.py @@ -1,57 +1,83 @@ -import narwhals as nw +import re + import pandas as pd import polars as pl import pytest from feature_engine.imputation import CategoricalImputer +from tests.backend_helpers import frame_to_dict, null_count -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], -} +# init parameters +@pytest.mark.parametrize( + "imputation_method", + ["arbitrary", "mean", 1, None, ("missing",), ["frequent"]], +) +def test_error_when_imputation_method_not_frequent_or_missing(imputation_method): + msg = ( + "imputation_method takes only values 'missing' or 'frequent'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method=imputation_method) -def _cols(X, columns): - # to_dict(as_series=False) is a convenient, backend-agnostic way to read - # values back out for comparison, regardless of pandas vs polars. - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - return {c: result[c] for c in columns} +@pytest.mark.parametrize( + "ignore_format", + [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], +) +def test_error_when_ignore_format_is_not_boolean(ignore_format): + msg = ( + "ignore_format takes only booleans True and False. " + f"Got {ignore_format} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method="missing", ignore_format=ignore_format) -def _null_count(X, col): - return nw.from_native(X, eager_only=True)[col].null_count() +@pytest.mark.parametrize( + "return_object", + [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], +) +def test_error_when_return_object_is_not_boolean(return_object): + msg = ( + "return_object takes only booleans True and False. " + f"Got {return_object} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + CategoricalImputer(imputation_method="missing", return_object=return_object) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_impute_with_string_missing_and_automatically_find_variables(make_df): - df_na = make_df(DATA) - imputer = CategoricalImputer(imputation_method="missing", variables=None) - X_transformed = imputer.fit_transform(df_na) - # test init params - assert imputer.imputation_method == "missing" - assert imputer.variables is None +@pytest.mark.parametrize( + "imputation_method, fill_value, return_object, ignore_format", + [ + ("missing", "Missing", False, False), + ("missing", 0, True, True), + ("frequent", "Unknown", False, True), + ("frequent", 1.5, True, False), + ], +) +def test_init_param_assignment( + imputation_method, fill_value, return_object, ignore_format +): + imputer = CategoricalImputer( + imputation_method=imputation_method, + fill_value=fill_value, + return_object=return_object, + ignore_format=ignore_format, + ) + assert imputer.imputation_method == imputation_method + assert imputer.fill_value == fill_value + assert imputer.return_object is return_object + assert imputer.ignore_format is ignore_format + + +# fit and transform +def test_impute_with_string_missing_and_automatically_find_variables( + make_df, data_na +): + imputer = CategoricalImputer(imputation_method="missing", variables=None) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies"] @@ -65,38 +91,31 @@ def test_impute_with_string_missing_and_automatically_find_variables(make_df): # test transform output # selected columns should have no NA # non selected columns should still have NA - assert _null_count(X_transformed, "Name") == 0 - assert _null_count(X_transformed, "City") == 0 - assert _null_count(X_transformed, "Studies") == 0 - assert _null_count(X_transformed, "Age") > 0 - assert _null_count(X_transformed, "Marks") > 0 - assert _cols(X_transformed, ["Name", "City", "Studies"]) == { - "Name": [ - "tom", "nick", "krish", "Missing", "peter", "Missing", "fred", "sam", - ], - "City": [ - "London", "Manchester", "Missing", "Missing", "London", "London", - "Bristol", "Manchester", - ], - "Studies": [ - "Bachelor", "Bachelor", "Missing", "Missing", "Bachelor", "PhD", - "None", "Masters", - ], - } + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Name") == 0 + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Studies") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + result = frame_to_dict(X_transformed) + assert result["Name"] == [ + "tom", "nick", "krish", "Missing", "peter", "Missing", "fred", "sam", + ] + assert result["City"] == [ + "London", "Manchester", "Missing", "Missing", "London", "London", + "Bristol", "Manchester", + ] + assert result["Studies"] == [ + "Bachelor", "Bachelor", "Missing", "Missing", "Bachelor", "PhD", + "None", "Masters", + ] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_user_defined_string_and_automatically_find_variables(make_df): - df_na = make_df(DATA) +def test_user_defined_string_and_automatically_find_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="missing", fill_value="Unknown", variables=None ) - X_transformed = imputer.fit_transform(df_na) - - # test init params - assert imputer.imputation_method == "missing" - assert imputer.fill_value == "Unknown" - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies"] @@ -108,71 +127,62 @@ def test_user_defined_string_and_automatically_find_variables(make_df): } # test transform output - assert _null_count(X_transformed, "Name") == 0 - assert _null_count(X_transformed, "City") == 0 - assert _null_count(X_transformed, "Studies") == 0 - assert _null_count(X_transformed, "Age") > 0 - assert _null_count(X_transformed, "Marks") > 0 - assert _cols(X_transformed, ["City"]) == { - "City": [ - "London", "Manchester", "Unknown", "Unknown", "London", "London", - "Bristol", "Manchester", - ], - } + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Name") == 0 + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Studies") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + assert frame_to_dict(X_transformed)["City"] == [ + "London", "Manchester", "Unknown", "Unknown", "London", "London", + "Bristol", "Manchester", + ] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_mode_imputation_and_single_variable(make_df): - df_na = make_df(DATA) +def test_mode_imputation_and_single_variable(make_df, data_na): imputer = CategoricalImputer(imputation_method="frequent", variables="City") - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na)) - # test init, fit and transform params, attr and output - assert imputer.imputation_method == "frequent" - assert imputer.variables == "City" + # test fit attr and transform output assert imputer.variables_ == ["City"] assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"City": "London"} - assert _null_count(X_transformed, "City") == 0 - assert _null_count(X_transformed, "Age") > 0 - assert _null_count(X_transformed, "Marks") > 0 - assert _cols(X_transformed, ["City"]) == { - "City": [ - "London", "Manchester", "London", "London", "London", "London", - "Bristol", "Manchester", - ], - } + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + assert null_count(X_transformed, "Age") > 0 + assert null_count(X_transformed, "Marks") > 0 + assert frame_to_dict(X_transformed)["City"] == [ + "London", "Manchester", "London", "London", "London", "London", + "Bristol", "Manchester", + ] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_mode_imputation_with_multiple_variables(make_df): - df_na = make_df(DATA) +def test_mode_imputation_with_multiple_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="frequent", variables=["Studies", "City"] ) - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attr and transform output assert imputer.imputer_dict_ == {"Studies": "Bachelor", "City": "London"} - assert _cols(X_transformed, ["Studies", "City"]) == { - "Studies": [ - "Bachelor", "Bachelor", "Bachelor", "Bachelor", "Bachelor", "PhD", - "None", "Masters", - ], - "City": [ - "London", "Manchester", "London", "London", "London", "London", - "Bristol", "Manchester", - ], - } + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) + assert result["Studies"] == [ + "Bachelor", "Bachelor", "Bachelor", "Bachelor", "Bachelor", "PhD", + "None", "Masters", + ] + assert result["City"] == [ + "London", "Manchester", "London", "London", "London", "London", + "Bristol", "Manchester", + ] -def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical(): - # Backend-specific: casting a numeric column to pandas' "object" dtype - # while keeping numeric values (Option 1 in the docstring) is a pandas - # dtype quirk with no polars equivalent - polars stays typed, so - # fillna+infer_objects' auto-revert-to-numeric never happens there - # (see test_polars_return_object_is_a_no_op below). - df_na = pd.DataFrame(DATA) +def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical( + data_na, +): + # casting a numeric column to pandas' "object" dtype while keeping + # numeric values is a pandas quirk with no polars equivalent. + df_na = pd.DataFrame(data_na) df_na["Marks"] = df_na["Marks"].astype("O") imputer = CategoricalImputer( imputation_method="frequent", variables=["City", "Studies", "Marks"] @@ -183,7 +193,6 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical() X_reference["Marks"] = X_reference["Marks"].astype(float).fillna(0.8) X_reference["City"] = X_reference["City"].fillna("London") X_reference["Studies"] = X_reference["Studies"].fillna("Bachelor") - assert imputer.variables == ["City", "Studies", "Marks"] assert imputer.variables_ == ["City", "Studies", "Marks"] assert imputer.imputer_dict_ == { "Studies": "Bachelor", @@ -194,10 +203,11 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_numerical() pd.testing.assert_frame_equal(X_transformed, X_reference) -def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object(): - # Backend-specific: see comment on the test above - return_object only - # has an effect on pandas, where infer_objects() silently upcasts. - df_na = pd.DataFrame(DATA) +def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object( + data_na, +): + # pandas only: see comment on the test above. + df_na = pd.DataFrame(data_na) df_na["Marks"] = df_na["Marks"].astype("O") imputer = CategoricalImputer( imputation_method="frequent", @@ -209,9 +219,7 @@ def test_imputation_of_numerical_vars_cast_as_object_and_returned_as_object(): def test_polars_return_object_is_a_no_op(): - # Documents the backend difference: polars never silently upcasts a - # String-typed column back to numeric (no infer_objects equivalent), - # so return_object has nothing to do there, unlike on pandas above. + # polars never casts String back to numeric, so return_object has no effect df_na = pl.DataFrame( {"Marks": ["0.9", "0.8", "0.7", None, "0.3", None, "0.8", "0.6"]} ) @@ -225,23 +233,18 @@ def test_polars_return_object_is_a_no_op(): assert X_transformed.schema["Marks"] == pl.String -def test_error_when_imputation_method_not_frequent_or_missing(): - with pytest.raises(ValueError): - CategoricalImputer(imputation_method="arbitrary") - - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_uses_smallest_mode_when_variable_has_multiple_modes(make_df): +def test_uses_smallest_mode_when_variable_has_multiple_modes(make_df, data_na): # every non-null value of "Name" is unique, so all are modes. The imputer - # picks the sorted-smallest one ("fred") - deterministically and - # identically for pandas and polars - instead of raising. - df_na = make_df(DATA) + # picks the sorted-smallest one ("fred") deterministically. + df_na = make_df(data_na) # explicit variable imputer = CategoricalImputer(imputation_method="frequent", variables="Name") imputer.fit(df_na) assert imputer.imputer_dict_ == {"Name": "fred"} - assert _cols(imputer.transform(df_na), ["Name"])["Name"] == [ + X_transformed = imputer.transform(df_na) + assert isinstance(X_transformed, make_df) + assert frame_to_dict(X_transformed)["Name"] == [ "tom", "nick", "krish", @@ -252,50 +255,40 @@ def test_uses_smallest_mode_when_variable_has_multiple_modes(make_df): "sam", ] - # auto-selected: only "Name" is multi-mode; "City" and "Studies" each have - # a single mode and are unaffected. + # auto-selected: only "Name" is multi-mode; "City" has + # a single mode and is unaffected. imputer = CategoricalImputer(imputation_method="frequent") imputer.fit(df_na) assert imputer.imputer_dict_["Name"] == "fred" assert imputer.imputer_dict_["City"] == "London" -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_impute_numerical_variables(make_df): - df_na = make_df(DATA) +def test_impute_numerical_variables(make_df, data_na): imputer = CategoricalImputer( imputation_method="missing", fill_value=0, variables=["Name", "City", "Studies", "Age", "Marks"], ignore_format=True, ) - X_transformed = imputer.fit_transform(df_na) - - # test init params - assert imputer.imputation_method == "missing" - assert imputer.variables == ["Name", "City", "Studies", "Age", "Marks"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 5 # test transform params: no nulls left anywhere + assert isinstance(X_transformed, make_df) for col in ["Name", "City", "Studies", "Age", "Marks"]: - assert _null_count(X_transformed, col) == 0 + assert null_count(X_transformed, col) == 0 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_impute_numerical_variables_with_mode(make_df): - df_na = make_df(DATA) +def test_impute_numerical_variables_with_mode(make_df, data_na): imputer = CategoricalImputer( imputation_method="frequent", variables=["City", "Studies", "Marks"], ignore_format=True, ) - X_transformed = imputer.fit_transform(df_na) - - # test init params - assert imputer.variables == ["City", "Studies", "Marks"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["City", "Studies", "Marks"] @@ -307,17 +300,14 @@ def test_impute_numerical_variables_with_mode(make_df): } # test transform output + assert isinstance(X_transformed, make_df) for col in ["City", "Studies", "Marks"]: - assert _null_count(X_transformed, col) == 0 + assert null_count(X_transformed, col) == 0 -def test_variables_cast_as_category_missing(): - # Backend-specific: pandas' category dtype needs an explicit - # cat.add_categories() step before fillna, or it raises TypeError - - # polars' Categorical widens itself automatically on fill_null (see - # test_polars_categorical_dtype_widens_on_missing_fill below), so - # there is no shared behaviour to parametrize here. - df_na = pd.DataFrame(DATA) +def test_variables_cast_as_category_missing(data_na): + # pandas only + df_na = pd.DataFrame(data_na) df_na["City"] = df_na["City"].astype("category") imputer = CategoricalImputer(imputation_method="missing", variables=None) @@ -341,12 +331,9 @@ def test_variables_cast_as_category_missing(): pd.testing.assert_frame_equal(X_transformed, X_reference) -def test_variables_cast_as_category_frequent(): - # Backend-specific: see comment on test_variables_cast_as_category_missing. - # The frequent-mode fill value is always an existing category, so this - # particular case wouldn't actually exercise a real pandas-vs-polars - # difference - it is kept pandas-only to match the "missing" test above. - df_na = pd.DataFrame(DATA) +def test_variables_cast_as_category_frequent(data_na): + # pandas only + df_na = pd.DataFrame(data_na) df_na["City"] = df_na["City"].astype("category") df_na = df_na.drop(columns=["Name"]) # this variable has no mode @@ -367,11 +354,9 @@ def test_variables_cast_as_category_frequent(): pd.testing.assert_frame_equal(X_transformed, X_reference) -def test_polars_categorical_dtype_widens_on_missing_fill(): - # Correctness risk called out for this migration: polars' Categorical - # (unlike pandas' category dtype) accepts a brand-new value directly on - # fill_null - no add_categories-equivalent step is needed. - df_na = pl.DataFrame(DATA).with_columns(pl.col("City").cast(pl.Categorical)) +def test_polars_categorical_dtype_widens_on_missing_fill(data_na): + # polars only. + df_na = pl.DataFrame(data_na).with_columns(pl.col("City").cast(pl.Categorical)) imputer = CategoricalImputer( imputation_method="missing", fill_value="Missing", variables=["City"] @@ -379,20 +364,17 @@ def test_polars_categorical_dtype_widens_on_missing_fill(): X_transformed = imputer.fit_transform(df_na) assert X_transformed.schema["City"] == pl.Categorical - assert X_transformed["City"].null_count() == 0 - assert X_transformed["City"].to_list() == [ + assert null_count(X_transformed, "City") == 0 + assert frame_to_dict(X_transformed)["City"] == [ "London", "Manchester", "Missing", "Missing", "London", "London", "Bristol", "Manchester", ] -def test_polars_enum_fixed_categories_raises_on_missing_fill(): - # Correctness risk called out for this migration: polars' Enum has a - # *fixed* category set. Filling with a value outside it would otherwise - # silently write null (no error) instead of the intended fill value - - # we raise a clear error instead of corrupting data silently. +def test_polars_enum_fixed_categories_raises_on_missing_fill(data_na): + # polars only. enum_dtype = pl.Enum(["London", "Manchester", "Bristol"]) - df_na = pl.DataFrame(DATA).with_columns(pl.col("City").cast(enum_dtype)) + df_na = pl.DataFrame(data_na).with_columns(pl.col("City").cast(enum_dtype)) imputer = CategoricalImputer( imputation_method="missing", fill_value="Missing", variables=["City"] @@ -405,14 +387,4 @@ def test_polars_enum_fixed_categories_raises_on_missing_fill(): imputation_method="missing", fill_value="London", variables=["City"] ) X_transformed = imputer_ok.fit_transform(df_na) - assert X_transformed["City"].null_count() == 0 - - -@pytest.mark.parametrize( - "ignore_format", - [22.3, 1, "HOLA", {"key1": "value1", "key2": "value2", "key3": "value3"}], -) -def test_error_when_ignore_format_is_not_boolean(ignore_format): - msg = "ignore_format takes only booleans True and False" - with pytest.raises(ValueError, match=msg): - CategoricalImputer(imputation_method="missing", ignore_format=ignore_format) + assert null_count(X_transformed, "City") == 0 diff --git a/tests/test_imputation/test_drop_missing_data.py b/tests/test_imputation/test_drop_missing_data.py index b08d21eb0..715fa898f 100644 --- a/tests/test_imputation/test_drop_missing_data.py +++ b/tests/test_imputation/test_drop_missing_data.py @@ -1,130 +1,97 @@ -import datetime as dt +import re -import narwhals as nw -import pandas as pd -import polars as pl import pytest from feature_engine.imputation import DropMissingData +from tests.backend_helpers import frame_to_dict, make_series, null_count -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], - # never null: exercises a datetime variable that missing_only=True - # should exclude from variables_ (it never contributes NA). - "dob": [dt.datetime(2020, 2, 24, 0, i) for i in range(8)], -} - - -def _cols(X, columns): - # to_dict(as_series=False) is a convenient, backend-agnostic way to read - # values back out for comparison, regardless of pandas vs polars. pandas - # represents missing numerics as float nan, not None, so normalize nan - # to None to compare uniformly across backends. - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - return { - c: [None if isinstance(v, float) and v != v else v for v in result[c]] - for c in columns - } - - -def _to_list(y): - return nw.from_native(y, series_only=True).to_list() - - -def _make_series(make_df, values): - return pd.Series(values) if make_df is pd.DataFrame else pl.Series(values) - - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_detect_variables_with_na(make_df): - df_na = make_df(DATA) + +# init parameters +@pytest.mark.parametrize("missing_only", ["missing_only", 1, None]) +def test_error_when_missing_only_not_bool(missing_only): + msg = f"missing_only takes values True or False. Got {missing_only} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropMissingData(missing_only=missing_only) + + +@pytest.mark.parametrize("threshold", [1.01, -0.01, 0, "0.5"]) +def test_error_when_threshold_not_between_0_and_1(threshold): + msg = f"threshold must be a value between 0 < x <= 1. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropMissingData(threshold=threshold) + + +@pytest.mark.parametrize( + "missing_only, threshold", + [(True, None), (False, None), (True, 0.5), (False, 1)], +) +def test_init_param_assignment(missing_only, threshold): + imputer = DropMissingData(missing_only=missing_only, threshold=threshold) + assert imputer.missing_only is missing_only + assert imputer.threshold == threshold + + +# fit and transform +def test_detect_variables_with_na(make_df, data_na_dob): # test case 1: automatically detect variables with missing data imputer = DropMissingData(missing_only=True, variables=None) - X_transformed = imputer.fit_transform(df_na) - # init params - assert imputer.missing_only is True - assert imputer.threshold is None - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na_dob)) # fit params assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 6 # transform outputs: only rows complete in variables_ survive + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (5, 6) - assert _cols(X_transformed, ["Age"]) == {"Age": [20, 21, 23, 41, 37]} + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 23, 41, 37] for var in imputer.variables_: - assert nw.from_native(X_transformed, eager_only=True)[var].null_count() == 0 + assert null_count(X_transformed, var) == 0 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_transform_x_y(make_df): - df_na = make_df(DATA) - y = _make_series(make_df, list(range(8))) +def test_transform_x_y(make_df, data_na_dob): + df_na = make_df(data_na_dob) + y = make_series(make_df, list(range(8))) imputer = DropMissingData(missing_only=True, variables=None) X_transformed = imputer.fit_transform(df_na) assert X_transformed.shape == (5, 6) assert len(X_transformed) != len(y) Xt, yt = imputer.transform_x_y(df_na, y) + assert isinstance(Xt, make_df) + assert isinstance(yt, type(y)) # rows 0, 1, 4, 6, 7 are the ones complete in Name/City/Studies/Age/Marks - assert _to_list(yt) == [0, 1, 4, 6, 7] - assert _cols(Xt, ["Age"]) == {"Age": [20, 21, 23, 41, 37]} + assert list(yt) == [0, 1, 4, 6, 7] + assert frame_to_dict(Xt)["Age"] == [20, 21, 23, 41, 37] assert len(Xt) == len(yt) assert len(df_na) != len(Xt) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_selelct_all_variables_when_variables_is_none(make_df): - df_na = make_df(DATA) +def test_selelct_all_variables_when_variables_is_none(make_df, data_na_dob): imputer = DropMissingData(missing_only=False, variables=None) - X_transformed = imputer.fit_transform(df_na) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.n_features_in_ == 6 assert imputer.variables_ == [ "Name", "City", "Studies", "Age", "Marks", "dob" ] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (5, 6) for var in imputer.variables_: - assert nw.from_native(X_transformed, eager_only=True)[var].null_count() == 0 + assert null_count(X_transformed, var) == 0 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_detect_variables_with_na_in_variables_entered_by_user(make_df): - df_na = make_df(DATA) +def test_detect_variables_with_na_in_variables_entered_by_user(make_df, data_na_dob): imputer = DropMissingData( missing_only=True, variables=["City", "Studies", "Age", "dob"] ) - X_transformed = imputer.fit_transform(df_na) - assert imputer.variables == ["City", "Studies", "Age", "dob"] + X_transformed = imputer.fit_transform(make_df(data_na_dob)) # dob never has NA in the train set, so it's dropped from variables_ assert imputer.variables_ == ["City", "Studies", "Age"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (6, 6) - assert _cols(X_transformed, ["Age"]) == {"Age": [20, 21, 23, 40, 41, 37]} + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 23, 40, 41, 37] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_return_na_data_method(make_df): - df_na = make_df(DATA) +def test_return_na_data_method(make_df, data_na_dob): + df_na = make_df(data_na_dob) # test with vars and threshold: return_na_data must return the exact # complement of transform() - row 2 has 2 of 4 variables present, which @@ -135,22 +102,23 @@ def test_return_na_data_method(make_df): ) imputer.fit_transform(df_na) X_nona = imputer.return_na_data(df_na) + assert isinstance(X_nona, make_df) assert X_nona.shape[0] == 1 - assert _cols(X_nona, ["Age"]) == {"Age": [None]} + assert frame_to_dict(X_nona)["Age"] == [None] # test without vars & threshold imputer = DropMissingData() imputer.fit_transform(df_na) X_nona = imputer.return_na_data(df_na) + assert isinstance(X_nona, make_df) assert X_nona.shape[0] == 3 - assert _cols(X_nona, ["Age"]) == {"Age": [19, None, 40]} + assert frame_to_dict(X_nona)["Age"] == [19, None, 40] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_transform_and_return_na_data_partition_input(make_df): +def test_transform_and_return_na_data_partition_input(make_df, data_na_dob): # transform() (rows kept) and return_na_data() (rows dropped) must # partition the input exactly: no row in both, no row in neither. - df_na = make_df(DATA) + df_na = make_df(data_na_dob) for threshold in [None, 1, 0.75, 0.5, 0.25, 0.01]: imputer = DropMissingData( threshold=threshold, variables=["City", "Studies", "Age", "Marks"] @@ -159,79 +127,62 @@ def test_transform_and_return_na_data_partition_input(make_df): kept = imputer.transform(df_na) dropped = imputer.return_na_data(df_na) assert kept.shape[0] + dropped.shape[0] == df_na.shape[0] - kept_age = set(_cols(kept, ["Age"])["Age"]) - dropped_age = set(_cols(dropped, ["Age"])["Age"]) + kept_age = set(frame_to_dict(kept)["Age"]) + dropped_age = set(frame_to_dict(dropped)["Age"]) assert kept_age.isdisjoint(dropped_age) -def test_error_when_missing_only_not_bool(): - with pytest.raises(ValueError): - DropMissingData(missing_only="missing_only") - - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_threshold(make_df): - df_na = make_df(DATA) +def test_threshold(make_df, data_na_dob): + df_na = make_df(data_na_dob) # Each row must have 100% data available imputer = DropMissingData(threshold=1) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 23, 41, 37]} + assert isinstance(X, make_df) + assert frame_to_dict(X)["Age"] == [20, 21, 23, 41, 37] # Each row must have at least 1% data available imputer = DropMissingData(threshold=0.01) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 19, None, 23, 40, 41, 37]} + assert frame_to_dict(X)["Age"] == [20, 21, 19, None, 23, 40, 41, 37] # Each row must have at least 50% data available imputer = DropMissingData(threshold=0.50) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 19, 23, 40, 41, 37]} + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 40, 41, 37] # threshold overrides missing_only, so the same 3 checks hold verbatim # with missing_only=False: imputer = DropMissingData(threshold=1, missing_only=False) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 23, 41, 37]} + assert frame_to_dict(X)["Age"] == [20, 21, 23, 41, 37] imputer = DropMissingData(threshold=0.01, missing_only=False) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 19, None, 23, 40, 41, 37]} + assert frame_to_dict(X)["Age"] == [20, 21, 19, None, 23, 40, 41, 37] imputer = DropMissingData(threshold=0.50, missing_only=False) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 19, 23, 40, 41, 37]} - - -def test_threshold_value_error(): - with pytest.raises(ValueError): - DropMissingData(threshold=1.01) - - with pytest.raises(ValueError): - DropMissingData(threshold=-0.01) - - with pytest.raises(ValueError): - DropMissingData(threshold=0) + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 40, 41, 37] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_threshold_with_variables(make_df): - df_na = make_df(DATA) +def test_threshold_with_variables(make_df, data_na_dob): + df_na = make_df(data_na_dob) # Each row must have 100% data available for column ['Marks'] imputer = DropMissingData(threshold=1, variables=["Marks"]) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 19, 23, 41, 37]} + assert isinstance(X, make_df) + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 41, 37] # Each row must have 75% data available for ['City', 'Studies', 'Age', 'Marks'] imputer = DropMissingData( threshold=0.75, variables=["City", "Studies", "Age", "Marks"] ) X = imputer.fit_transform(df_na) - assert _cols(X, ["Age"]) == {"Age": [20, 21, 23, 40, 41, 37]} + assert frame_to_dict(X)["Age"] == [20, 21, 23, 40, 41, 37] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) def test_missing_only_finds_no_variables_leaves_data_unchanged(make_df): # A clean training set has nothing for missing_only=True to select: # variables_ ends up empty, and transform()/return_na_data() must not @@ -241,6 +192,8 @@ def test_missing_only_finds_no_variables_leaves_data_unchanged(make_df): imputer = DropMissingData() Xt = imputer.fit_transform(X) assert imputer.variables_ == [] - assert Xt.shape == (3, 2) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == clean_data X_nona = imputer.return_na_data(X) + assert isinstance(X_nona, make_df) assert X_nona.shape == (0, 2) diff --git a/tests/test_imputation/test_end_tail_imputer.py b/tests/test_imputation/test_end_tail_imputer.py index 36a1db459..4a6d0e532 100644 --- a/tests/test_imputation/test_end_tail_imputer.py +++ b/tests/test_imputation/test_end_tail_imputer.py @@ -1,75 +1,59 @@ -import narwhals as nw +import re + import numpy as np -import pandas as pd -import polars as pl import pytest from feature_engine.imputation import EndTailImputer +from tests.backend_helpers import frame_to_dict, null_count + + +# init parameters +@pytest.mark.parametrize( + "imputation_method", ["arbitrary", "mean", 1, ("iqr",), ["iqr"]] +) +def test_error_when_imputation_method_is_not_permitted(imputation_method): + msg = ( + "imputation_method takes only values 'gaussian', 'iqr' or 'max'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(imputation_method=imputation_method) + + +@pytest.mark.parametrize("tail", ["arbitrary", "both", 1, ("right",), ["right"]]) +def test_error_when_tail_is_not_permitted(tail): + msg = f"tail takes only values 'right' or 'left'. Got {tail} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(tail=tail) + + +@pytest.mark.parametrize("fold", [-1, 0, -0.5, "3", None, [3], True]) +def test_error_when_fold_is_not_positive_number(fold): + msg = f"fold takes only positive numbers. Got {fold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + EndTailImputer(fold=fold) + + +@pytest.mark.parametrize( + "imputation_method, tail, fold", + [("gaussian", "right", 3), ("iqr", "left", 1.5), ("max", "right", 2)], +) +def test_init_param_assignment(imputation_method, tail, fold): + imputer = EndTailImputer(imputation_method=imputation_method, tail=tail, fold=fold) + assert imputer.imputation_method == imputation_method + assert imputer.tail == tail + assert imputer.fold == fold -# Missing values are written as `None`, not `np.nan`: polars treats np.nan as -# a real float value (not a null), so mean/std/quantile would NOT skip it, -# unlike pandas' NaN-as-missing default. `None` becomes a null on both -# backends and is skipped by both, keeping the two code paths comparable. -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], -} - - -def _none_to_nan(values): - # Missing values print as None for polars, NaN for pandas float columns - # - both mean "missing" here, so normalize both sides before comparing. - return [np.nan if v is None else v for v in values] - - -def assert_df_equal(X, expected: dict, abs_tol: float = 1e-5) -> None: - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - assert list(result.keys()) == list(expected.keys()) - for col, values in expected.items(): - assert _none_to_nan(result[col]) == pytest.approx( - _none_to_nan(values), abs=abs_tol, nan_ok=True - ) - - -def _missing_count(X, columns) -> int: - nw_X = nw.from_native(X, eager_only=True) - return sum(int(nw_X.get_column(c).is_null().sum()) for c in columns) - - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_automatically_find_variables_and_gaussian_imputation_on_right_tail(make_df): - df = make_df(DATA) + +# fit and transform +def test_automatically_find_variables_and_gaussian_imputation_on_right_tail( + make_df, data_na +): imputer = EndTailImputer( imputation_method="gaussian", tail="right", fold=3, variables=None ) - X_transformed = imputer.fit_transform(df) + X_transformed = imputer.fit_transform(make_df(data_na)) - # test init params - assert imputer.imputation_method == "gaussian" - assert imputer.tail == "right" - assert imputer.fold == 3 - assert imputer.variables is None # test fit attr assert imputer.variables_ == ["Age", "Marks"] assert imputer.n_features_in_ == 5 @@ -77,76 +61,60 @@ def test_automatically_find_variables_and_gaussian_imputation_on_right_tail(make assert rounded == {"Age": 58.949, "Marks": 1.324} # transform output: indicated vars ==> no NA, not indicated vars with NA - assert _missing_count(X_transformed, ["Age", "Marks"]) == 0 - assert _missing_count(X_transformed, ["City", "Name"]) > 0 - - expected = dict(DATA) - expected["Age"] = [20, 21, 19, 58.94908118478389, 23, 40, 41, 37] - expected["Marks"] = [ - 0.9, 0.8, 0.7, 1.3244261503263175, 0.3, 1.3244261503263175, 0.8, 0.6, - ] - assert_df_equal(X_transformed, expected) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "City") > 0 + assert null_count(X_transformed, "Name") > 0 + + expected = dict(data_na) + expected["Age"] = pytest.approx([20, 21, 19, 58.94908118478389, 23, 40, 41, 37]) + expected["Marks"] = pytest.approx( + [0.9, 0.8, 0.7, 1.3244261503263175, 0.3, 1.3244261503263175, 0.8, 0.6] + ) + assert frame_to_dict(X_transformed) == expected -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_user_enters_variables_and_iqr_imputation_on_right_tail(make_df): - df = make_df(DATA) +def test_user_enters_variables_and_iqr_imputation_on_right_tail(make_df, data_na): imputer = EndTailImputer( imputation_method="iqr", tail="right", fold=1.5, variables=["Age", "Marks"] ) - X_transformed = imputer.fit_transform(df) + X_transformed = imputer.fit_transform(make_df(data_na)) assert imputer.imputer_dict_ == {"Age": 65.5, "Marks": 1.0625} - assert _missing_count(X_transformed, ["Age", "Marks"]) == 0 + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 - expected = dict(DATA) - expected["Age"] = [20, 21, 19, 65.5, 23, 40, 41, 37] - expected["Marks"] = [0.9, 0.8, 0.7, 1.0625, 0.3, 1.0625, 0.8, 0.6] - assert_df_equal(X_transformed, expected) + expected = dict(data_na) + expected["Age"] = pytest.approx([20, 21, 19, 65.5, 23, 40, 41, 37]) + expected["Marks"] = pytest.approx([0.9, 0.8, 0.7, 1.0625, 0.3, 1.0625, 0.8, 0.6]) + assert frame_to_dict(X_transformed) == expected -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_user_enters_variables_and_max_value_imputation(make_df): - df = make_df(DATA) +def test_user_enters_variables_and_max_value_imputation(make_df, data_na): imputer = EndTailImputer( imputation_method="max", tail="right", fold=2, variables=["Age", "Marks"] ) - imputer.fit(df) + imputer.fit(make_df(data_na)) assert imputer.imputer_dict_ == {"Age": 82.0, "Marks": 1.8} -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_automatically_select_variables_and_gaussian_imputation_on_left_tail(make_df): - df = make_df(DATA) +def test_automatically_select_variables_and_gaussian_imputation_on_left_tail( + make_df, data_na +): imputer = EndTailImputer(imputation_method="gaussian", tail="left", fold=3) - imputer.fit(df) + imputer.fit(make_df(data_na)) rounded = {k: round(v, 3) for k, v in imputer.imputer_dict_.items()} assert rounded == {"Age": -1.521, "Marks": 0.042} -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_user_enters_variables_and_iqr_imputation_on_left_tail(make_df): - df = make_df(DATA) +def test_user_enters_variables_and_iqr_imputation_on_left_tail(make_df, data_na): imputer = EndTailImputer( imputation_method="iqr", tail="left", fold=1.5, variables=["Age", "Marks"] ) - imputer.fit(df) + imputer.fit(make_df(data_na)) assert imputer.imputer_dict_["Age"] == -6.5 assert np.round(imputer.imputer_dict_["Marks"], 3) == np.round( 0.36249999999999993, 3 ) - - -def test_error_when_imputation_method_is_not_permitted(): - with pytest.raises(ValueError, match="imputation_method takes only values"): - EndTailImputer(imputation_method="arbitrary") - - -def test_error_when_tail_is_string(): - with pytest.raises(ValueError, match="tail takes only values"): - EndTailImputer(tail="arbitrary") - - -def test_error_when_fold_is_1(): - with pytest.raises(ValueError, match="fold takes only positive numbers"): - EndTailImputer(fold=-1) diff --git a/tests/test_imputation/test_mean_median_imputer.py b/tests/test_imputation/test_mean_median_imputer.py index 6f4782a96..a3ca0dc33 100644 --- a/tests/test_imputation/test_mean_median_imputer.py +++ b/tests/test_imputation/test_mean_median_imputer.py @@ -1,11 +1,9 @@ import re -import narwhals as nw -import pandas as pd -import polars as pl import pytest from feature_engine.imputation import MeanImputer, MeanMedianImputer +from tests.backend_helpers import frame_to_dict, null_count DEPRECATION_WARNING = ( "MeanMedianImputer was deprecated in favour of MeanImputer in version " @@ -13,43 +11,6 @@ "use MeanImputer instead." ) -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], -} - - -def _cols(X, columns): - # to_dict(as_series=False) is a convenient, backend-agnostic way to read - # values back out for comparison, regardless of pandas vs polars. - result = nw.from_native(X, eager_only=True).to_dict(as_series=False) - return {c: result[c] for c in columns} - - -def _null_count(X, col): - return nw.from_native(X, eager_only=True)[col].null_count() - @pytest.fixture( params=[MeanImputer, MeanMedianImputer], @@ -66,20 +27,31 @@ def make_imputer(imputer_class, **kwargs): return imputer_class(**kwargs) -def test_mean_median_imputer_raises_future_warning(): - with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): - MeanMedianImputer() +# init parameters +@pytest.mark.parametrize( + "imputation_method", ["arbitrary", "mode", 1, None, ("mean",), ["median"]] +) +def test_error_with_wrong_imputation_method(imputer_class, imputation_method): + msg = ( + "imputation_method takes only values 'median' or 'mean'. " + f"Got {imputation_method} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + make_imputer(imputer_class, imputation_method=imputation_method) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_mean_imputation_and_automatically_select_variables(make_df, imputer_class): - df_na = make_df(DATA) - imputer = make_imputer(imputer_class, imputation_method="mean", variables=None) - X_transformed = imputer.fit_transform(df_na) +@pytest.mark.parametrize("imputation_method", ["mean", "median"]) +def test_init_param_assignment(imputer_class, imputation_method): + imputer = make_imputer(imputer_class, imputation_method=imputation_method) + assert imputer.imputation_method == imputation_method - # test init params - assert imputer.imputation_method == "mean" - assert imputer.variables is None + +# fit and transform +def test_mean_imputation_and_automatically_select_variables( + make_df, data_na, imputer_class +): + imputer = make_imputer(imputer_class, imputation_method="mean", variables=None) + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.variables_ == ["Age", "Marks"] @@ -92,11 +64,12 @@ def test_mean_imputation_and_automatically_select_variables(make_df, imputer_cla # test transform output: # selected variables should have no NA # not selected variables should still have NA - assert _null_count(X_transformed, "Age") == 0 - assert _null_count(X_transformed, "Marks") == 0 - assert _null_count(X_transformed, "Name") > 0 - assert _null_count(X_transformed, "City") > 0 - result = _cols(X_transformed, ["Age", "Marks"]) + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert null_count(X_transformed, "Marks") == 0 + assert null_count(X_transformed, "Name") > 0 + assert null_count(X_transformed, "City") > 0 + result = frame_to_dict(X_transformed) assert result["Age"] == pytest.approx( [20, 21, 19, 28.714285714285715, 23, 40, 41, 37] ) @@ -105,28 +78,24 @@ def test_mean_imputation_and_automatically_select_variables(make_df, imputer_cla ) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_median_imputation_when_user_enters_single_variables(make_df, imputer_class): - df_na = make_df(DATA) +def test_median_imputation_when_user_enters_single_variables( + make_df, data_na, imputer_class +): imputer = make_imputer( imputer_class, imputation_method="median", variables=["Age"] ) - X_transformed = imputer.fit_transform(df_na) - - # test init params - assert imputer.imputation_method == "median" - assert imputer.variables == ["Age"] + X_transformed = imputer.fit_transform(make_df(data_na)) # test fit attributes assert imputer.n_features_in_ == 5 assert imputer.imputer_dict_ == {"Age": 23.0} # test transform output - assert _null_count(X_transformed, "Age") == 0 - result = _cols(X_transformed, ["Age"]) - assert result["Age"] == [20, 21, 19, 23.0, 23, 40, 41, 37] + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "Age") == 0 + assert frame_to_dict(X_transformed)["Age"] == [20, 21, 19, 23.0, 23, 40, 41, 37] -def test_error_with_wrong_imputation_method(imputer_class): - with pytest.raises(ValueError): - make_imputer(imputer_class, imputation_method="arbitrary") +def test_mean_median_imputer_raises_future_warning(): + with pytest.warns(FutureWarning, match=re.escape(DEPRECATION_WARNING)): + MeanMedianImputer() diff --git a/tests/test_imputation/test_missing_indicator.py b/tests/test_imputation/test_missing_indicator.py index b7fdaaca2..70cf224c8 100644 --- a/tests/test_imputation/test_missing_indicator.py +++ b/tests/test_imputation/test_missing_indicator.py @@ -1,94 +1,60 @@ -import datetime +import re import warnings -import narwhals as nw import numpy as np import pandas as pd -import polars as pl import pytest - from sklearn.pipeline import Pipeline -from feature_engine.imputation import MissingIndicator, AddMissingIndicator - -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], - "dob": [ - datetime.datetime(2020, 2, 24) + datetime.timedelta(minutes=i) - for i in range(8) - ], -} - - -def _cols(X): - return list(nw.from_native(X, eager_only=True).columns) - - -def _col_sum(X, col): - return sum(nw.from_native(X, eager_only=True).get_column(col).to_list()) - - -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +from feature_engine.imputation import AddMissingIndicator, MissingIndicator +from tests.backend_helpers import frame_to_dict + +INDICATORS = [MissingIndicator, AddMissingIndicator] + + +# init parameters +@pytest.mark.parametrize("indicator_cls", INDICATORS) +@pytest.mark.parametrize("missing_only", ["missing_only", 1, None]) +def test_error_when_missing_only_not_bool(indicator_cls, missing_only): + msg = f"missing_only takes values True or False. Got {missing_only} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + indicator_cls(missing_only=missing_only) + + +@pytest.mark.parametrize("indicator_cls", INDICATORS) +@pytest.mark.parametrize("missing_only", [True, False]) +def test_init_param_assignment(indicator_cls, missing_only): + imputer = indicator_cls(missing_only=missing_only) + assert imputer.missing_only is missing_only + + +# fit and transform +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_detect_variables_with_missing_data_when_variables_is_none( - make_df, indicator_cls + make_df, data_na_dob, indicator_cls ): - X = make_df(DATA) # test case 1: automatically detect variables with missing data imputer = indicator_cls(missing_only=True, variables=None) - X_transformed = imputer.fit_transform(X) - - # init params - assert imputer.missing_only is True - assert imputer.variables is None + X_transformed = imputer.fit_transform(make_df(data_na_dob)) # fit params assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 6 # transform outputs + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 11) - assert "Name_na" in _cols(X_transformed) - assert _col_sum(X_transformed, "Name_na") == 2 + result = frame_to_dict(X_transformed) + assert "Name_na" in result + assert sum(result["Name_na"]) == 2 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_add_indicators_to_all_variables_when_variables_is_none( - make_df, indicator_cls + make_df, data_na_dob, indicator_cls ): - X = make_df(DATA) imputer = indicator_cls(missing_only=False, variables=None) - - X_transformed = imputer.fit_transform(X) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.variables_ == [ "Name", @@ -98,69 +64,49 @@ def test_add_indicators_to_all_variables_when_variables_is_none( "Marks", "dob", ] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 12) - assert "dob_na" in _cols(X_transformed) - assert _col_sum(X_transformed, "dob_na") == 0 + result = frame_to_dict(X_transformed) + assert "dob_na" in result + assert sum(result["dob_na"]) == 0 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_add_indicators_to_one_variable(make_df, indicator_cls): - X = make_df(DATA) +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_add_indicators_to_one_variable(make_df, data_na_dob, indicator_cls): imputer = indicator_cls(variables="Name") - - X_transformed = imputer.fit_transform(X) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) assert imputer.variables_ == ["Name"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 7) - assert "Name_na" in _cols(X_transformed) - assert _col_sum(X_transformed, "Name_na") == 2 + result = frame_to_dict(X_transformed) + assert "Name_na" in result + assert sum(result["Name_na"]) == 2 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_detect_variables_with_missing_data_in_variables_entered_by_user( - make_df, indicator_cls + make_df, data_na_dob, indicator_cls ): - X = make_df(DATA) imputer = indicator_cls( missing_only=True, variables=["City", "Studies", "Age", "dob"], ) + X_transformed = imputer.fit_transform(make_df(data_na_dob)) - X_transformed = imputer.fit_transform(X) - - assert imputer.variables == ["City", "Studies", "Age", "dob"] assert imputer.variables_ == ["City", "Studies", "Age"] + assert isinstance(X_transformed, make_df) assert X_transformed.shape == (8, 9) - assert "City_na" in _cols(X_transformed) - assert "dob_na" not in _cols(X_transformed) - assert _col_sum(X_transformed, "City_na") == 2 - - -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_error_when_missing_only_not_bool(indicator_cls): - with pytest.raises(ValueError): - indicator_cls(missing_only="missing_only") + result = frame_to_dict(X_transformed) + assert "City_na" in result + assert "dob_na" not in result + assert sum(result["City_na"]) == 2 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_get_feature_names_out(make_df, indicator_cls): - X = make_df(DATA) - original_features = _cols(X) +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_get_feature_names_out(make_df, data_na_dob, indicator_cls): + X = make_df(data_na_dob) + original_features = list(data_na_dob) tr = indicator_cls(missing_only=False) tr.fit(X) @@ -187,19 +133,12 @@ def test_get_feature_names_out(make_df, indicator_cls): tr.get_feature_names_out(["Name", "hola"]) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) -def test_get_feature_names_out_from_pipeline(make_df, indicator_cls): - X = make_df(DATA) - original_features = _cols(X) - - tr = Pipeline( - [("transformer", indicator_cls(missing_only=False))] - ) +@pytest.mark.parametrize("indicator_cls", INDICATORS) +def test_get_feature_names_out_from_pipeline(make_df, data_na_dob, indicator_cls): + X = make_df(data_na_dob) + original_features = list(data_na_dob) + tr = Pipeline([("transformer", indicator_cls(missing_only=False))]) tr.fit(X) out = [f + "_na" for f in original_features] @@ -209,13 +148,9 @@ def test_get_feature_names_out_from_pipeline(make_df, indicator_cls): assert tr.get_feature_names_out(input_features=original_features) == feat_out -@pytest.mark.parametrize( - "indicator_cls", - [MissingIndicator, AddMissingIndicator], -) +@pytest.mark.parametrize("indicator_cls", INDICATORS) def test_no_performance_warning_with_many_variables(indicator_cls): - # pandas-only: exercises the pandas fast path's PerformanceWarning - # behaviour specifically, not a cross-backend value comparison. + # pandas-only. n_cols = 101 df = pd.DataFrame( diff --git a/tests/test_imputation/test_random_sample_imputer.py b/tests/test_imputation/test_random_sample_imputer.py index e69de157a..feca6ca57 100644 --- a/tests/test_imputation/test_random_sample_imputer.py +++ b/tests/test_imputation/test_random_sample_imputer.py @@ -1,119 +1,116 @@ # Authors: Soledad Galli # License: BSD 3 clause -import narwhals as nw +import re + +import numpy as np import pandas as pd import polars as pl import pytest from feature_engine.imputation import RandomSampleImputer -from feature_engine.imputation.random_sample import _define_seed - -DATA = { - "Name": ["tom", "nick", "krish", None, "peter", None, "fred", "sam"], - "City": [ - "London", - "Manchester", - None, - None, - "London", - "London", - "Bristol", - "Manchester", - ], - "Studies": [ - "Bachelor", - "Bachelor", - None, - None, - "Bachelor", - "PhD", - "None", - "Masters", - ], - "Age": [20, 21, 19, None, 23, 40, 41, 37], - "Marks": [0.9, 0.8, 0.7, None, 0.3, None, 0.8, 0.6], -} +from feature_engine.imputation.random_sample import _hash_seeds +from tests.backend_helpers import frame_to_dict, null_count -def _null_count(X, col): - return nw.from_native(X, eager_only=True)[col].null_count() +# init parameters +@pytest.mark.parametrize( + "seed", ["arbitrary", "both", 1, None, ("general",), ["observation"]] +) +def test_error_if_seed_not_permitted_value(seed): + msg = f"seed takes only values 'general' or 'observation'. Got {seed} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed=seed) -def _values(X, col): - return nw.from_native(X, eager_only=True)[col].to_list() - - -def _pool(X, col): - # values available for the imputer to sample from, in the copy of the - # training data it stores at fit() - return set(nw.from_native(X, eager_only=True)[col].drop_nulls().to_list()) +@pytest.mark.parametrize("random_state", ["arbitrary", 0.5, ["Age"]]) +def test_error_if_random_state_not_integer_when_seed_is_general(random_state): + msg = ( + "if seed == 'general' then random_state must take an integer. " + f"Got {random_state} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed="general", random_state=random_state) -def _is_missing(v): - return v is None or (isinstance(v, float) and v != v) +@pytest.mark.parametrize("random_state", [None, [], ""]) +def test_error_if_random_state_is_empty_when_seed_is_observation(random_state): + msg = ( + "if seed == 'observation' the random state must take the name of one " + "or more variables which will be used to seed the imputer. " + f"Got {random_state} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RandomSampleImputer(seed="observation", random_state=random_state) -def _same_values(a, b): - # element-wise equality that treats None and float NaN as equal missing - # markers, since pandas' NaN and polars'/narwhals' None represent the - # same "missing" concept but compare unequal with plain `==`. - return len(a) == len(b) and all( - (_is_missing(x) and _is_missing(y)) or x == y for x, y in zip(a, b) +@pytest.mark.parametrize( + "random_state, seed", + [ + (None, "general"), + (5, "general"), + ("Age", "observation"), + (["Age", "Marks"], "observation"), + ], +) +def test_init_param_assignment(random_state, seed): + imputer = RandomSampleImputer(random_state=random_state, seed=seed) + assert imputer.random_state == random_state + assert imputer.seed == seed + + +# fit and transform +def test_hash_seeds(): + values = np.array( + [ + [25, 0.7], + [25.0, 0.7], + [0.0, 0.7], + [np.nan, 0.7], + [-0.0, 0.7], + [-30.0, 1e20], + ] ) + seeds = _hash_seeds(values) - -def test_define_seed(df_vartypes): - # _define_seed uses pandas' .loc label-based row access, so it is only - # ever called from the pandas branch of transform() - it is inherently - # pandas-only, unlike the rest of the transformer. - assert _define_seed(df_vartypes, 0, ["Age", "Marks"], how="add") == 21 - assert _define_seed(df_vartypes, 0, ["Age", "Marks"], how="multiply") == 18 - assert _define_seed(df_vartypes, 2, ["Age", "Marks"], how="add") == 20 - assert _define_seed(df_vartypes, 2, ["Age", "Marks"], how="multiply") == 13 - assert _define_seed(df_vartypes, 1, ["Age"], how="add") == 21 - assert _define_seed(df_vartypes, 3, ["Marks"], how="multiply") == 1 + # same values, same seed: ints and floats are equal, nan and -0.0 count as 0 + assert seeds[0] == seeds[1] + assert seeds[2] == seeds[3] == seeds[4] + assert seeds[0] != seeds[2] + # negative and large values give valid numpy seeds + assert all(0 <= seed < 2**32 for seed in seeds) + # the seed must not change between sessions or releases + assert _hash_seeds(np.array([[25.0, 0.7]]))[0] == 2067629302 -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_general_seed_plus_automatically_select_variables(make_df): - df_na = make_df(DATA) +def test_general_seed_plus_automatically_select_variables(make_df, data_na): + df_na = make_df(data_na) imputer = RandomSampleImputer(variables=None, random_state=5, seed="general") X_transformed = imputer.fit_transform(df_na) - # test init params - assert imputer.variables is None - assert imputer.random_state == 5 - assert imputer.seed == "general" - # test fit attrs assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 5 - for col in imputer.variables_: - assert _same_values(_values(imputer.X_, col), _values(df_na, col)) + assert frame_to_dict(imputer.X_) == frame_to_dict(df_na) - # no missing data left in any imputed variable + # no missing data left in any imputed variable, and every value used to + # fill NA came from the training data itself + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) for col in imputer.variables_: - assert _null_count(X_transformed, col) == 0 - # every value used to fill NA came from the training data itself - assert set(_values(X_transformed, col)) <= _pool(df_na, col) + assert null_count(X_transformed, col) == 0 + assert set(result[col]) <= {v for v in data_na[col] if v is not None} - # pandas' and narwhals/polars' sample() use different RNGs, so a fixed - # seed does not draw the same values across backends - only same seed + - # same backend is a reproducibility guarantee. Verify that guarantee. + # pandas and polars draw different values for the same seed, so we only check + # that the same seed on the same backend gives the same result. imputer2 = RandomSampleImputer(variables=None, random_state=5, seed="general") X_transformed2 = imputer2.fit_transform(df_na) - for col in imputer.variables_: - assert _values(X_transformed, col) == _values(X_transformed2, col) + assert frame_to_dict(X_transformed) == frame_to_dict(X_transformed2) def test_pandas_general_seed_reproduces_historic_values(df_na): - # Regression guard for the pandas fast-path specifically: transform()'s - # pandas branch is untouched code (still pandas' own .sample()/.loc), so - # for a fixed seed it must keep drawing the exact same values it drew - # before this narwhals migration. These literal values are inherently - # pandas-RNG-specific (see class docstring) and cannot be reproduced by - # any other backend, so this check is legitimately pandas-only. + # pandas only: with a fixed seed, pandas must return the same values as before + # the narwhals migration. polars uses a different random number generator. imputer = RandomSampleImputer(variables=None, random_state=5, seed="general") X_transformed = imputer.fit_transform(df_na) @@ -148,128 +145,158 @@ def test_pandas_general_seed_reproduces_historic_values(df_na): pd.testing.assert_frame_equal(X_transformed, ref, check_dtype=False) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_seed_per_observation_and_multiple_variables_in_random_state(make_df): - # Note: the variables used as seed should not have missing data, this I fill - data = dict(DATA) - data["Marks"] = [v if v is not None else 1 for v in data["Marks"]] - data["Age"] = [v if v is not None else 1 for v in data["Age"]] +def _data_without_na_in(data, columns): + # the variables used as seed should not have missing data + data = dict(data) + for col in columns: + data[col] = [v if v is not None else 1 for v in data[col]] + return data + + +@pytest.mark.parametrize("random_state", [["Marks", "Age"], "Age"]) +def test_seed_per_observation(make_df, data_na, random_state): + seed_vars = [random_state] if isinstance(random_state, str) else random_state + data = _data_without_na_in(data_na, seed_vars) df_na = make_df(data) imputer = RandomSampleImputer( - variables=["City", "Studies"], random_state=["Marks", "Age"], seed="observation" + variables=["City", "Studies"], + random_state=random_state, + seed="observation", ) X_transformed = imputer.fit_transform(df_na) - assert imputer.variables == ["City", "Studies"] - assert imputer.random_state == ["Marks", "Age"] - assert imputer.seed == "observation" + # fit() turns a single seeding variable name into a list + assert imputer.random_state == seed_vars + assert isinstance(X_transformed, make_df) + result = frame_to_dict(X_transformed) for col in ["City", "Studies"]: - assert _same_values(_values(imputer.X_, col), _values(df_na, col)) - assert _null_count(X_transformed, col) == 0 - assert set(_values(X_transformed, col)) <= _pool(df_na, col) + assert frame_to_dict(imputer.X_)[col] == data[col] + assert null_count(X_transformed, col) == 0 + assert set(result[col]) <= {v for v in data[col] if v is not None} # variables not selected for imputation are untouched - assert _same_values(_values(X_transformed, "Age"), _values(df_na, "Age")) + assert result["Age"] == data["Age"] # same seed, same backend -> same result imputer2 = RandomSampleImputer( - variables=["City", "Studies"], random_state=["Marks", "Age"], seed="observation" + variables=["City", "Studies"], + random_state=random_state, + seed="observation", ) X_transformed2 = imputer2.fit_transform(df_na) - for col in ["City", "Studies"]: - assert _values(X_transformed, col) == _values(X_transformed2, col) + assert frame_to_dict(X_transformed) == frame_to_dict(X_transformed2) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_seed_per_observation_plus_product_of_seeding_variables(make_df): - data = dict(DATA) - data["Marks"] = [v if v is not None else 1 for v in data["Marks"]] - data["Age"] = [v if v is not None else 1 for v in data["Age"]] - df_na = make_df(data) +DATA_SEED_TRAIN = { + "City": ["London", "Manchester", "Bristol", "Leeds", "York", "Bath", "Hull"], + "Age": [20.0, 21.0, 19.0, 23.0, 40.0, 41.0, 37.0], + "Marks": [0.9, 0.8, 0.7, 0.3, 0.6, 0.8, 0.5], +} +# rows 0 and 3 have identical seeding values and City missing +DATA_SEED_TEST = { + "City": [None, "Leeds", None, None, None], + "Age": [25.0, 30.0, 40.0, 25.0, 33.0], + "Marks": [0.7, 0.4, 0.6, 0.7, 0.2], +} + +def test_seed_per_observation_imputes_identical_rows_equally(make_df): imputer = RandomSampleImputer( - variables=["City", "Studies"], - random_state=["Marks", "Age"], - seed="observation", - seeding_method="multiply", + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) - X_transformed = imputer.fit_transform(df_na) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X_transformed = imputer.transform(make_df(DATA_SEED_TEST)) - assert imputer.variables == ["City", "Studies"] - assert imputer.random_state == ["Marks", "Age"] - assert imputer.seed == "observation" - for col in ["City", "Studies"]: - assert _same_values(_values(imputer.X_, col), _values(df_na, col)) - assert _null_count(X_transformed, col) == 0 - assert set(_values(X_transformed, col)) <= _pool(df_na, col) + city = frame_to_dict(X_transformed)["City"] + assert city[0] == city[3] - imputer2 = RandomSampleImputer( - variables=["City", "Studies"], - random_state=["Marks", "Age"], - seed="observation", - seeding_method="multiply", + +def test_seed_per_observation_does_not_depend_on_row_position(make_df): + imputer = RandomSampleImputer( + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) - X_transformed2 = imputer2.fit_transform(df_na) - for col in ["City", "Studies"]: - assert _values(X_transformed, col) == _values(X_transformed2, col) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df(DATA_SEED_TEST) + expected = frame_to_dict(imputer.transform(X))["City"] + # same rows in reverse order + X_reversed = make_df({k: v[::-1] for k, v in DATA_SEED_TEST.items()}) + reversed_city = frame_to_dict(imputer.transform(X_reversed))["City"] + assert reversed_city == expected[::-1] -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_seed_per_observation_with_only_1_variable_as_seed(make_df): - data = dict(DATA) - data["Age"] = [v if v is not None else 1 for v in data["Age"]] - df_na = make_df(data) + # each row imputed on its own + for i in range(len(expected)): + row_city = frame_to_dict(imputer.transform(X[i:i + 1]))["City"] + assert row_city == [expected[i]] + +def test_seed_per_observation_with_negative_and_large_seeding_values(make_df): imputer = RandomSampleImputer( - variables=["City", "Studies"], random_state="Age", seed="observation" + variables=["City"], random_state=["Age", "Marks"], seed="observation" ) - X_transformed = imputer.fit_transform(df_na) - - assert imputer.random_state == ["Age"] - for col in ["City", "Studies"]: - assert _same_values(_values(imputer.X_, col), _values(df_na, col)) - assert _null_count(X_transformed, col) == 0 - assert set(_values(X_transformed, col)) <= _pool(df_na, col) - - imputer2 = RandomSampleImputer( - variables=["City", "Studies"], random_state="Age", seed="observation" + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df( + { + "City": [None, None, "Leeds"], + "Age": [-30.0, 1e20, 20.0], + "Marks": [0.1, 1e20, 0.9], + } ) - X_transformed2 = imputer2.fit_transform(df_na) - for col in ["City", "Studies"]: - assert _values(X_transformed, col) == _values(X_transformed2, col) - + X_transformed = imputer.transform(X) -def test_error_if_seed_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="arbitrary") + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + assert set(frame_to_dict(X_transformed)["City"]) <= set(DATA_SEED_TRAIN["City"]) -def test_error_if_seeding_method_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seeding_method="arbitrary") +def test_seed_per_observation_uses_values_before_imputation_with_missing_as_zero( + make_df, +): + # Age is imputed and also seeds City: row 0 (Age missing) must seed like + # row 1 (Age 0), not with its imputed Age. + imputer = RandomSampleImputer( + variables=["Age", "City"], random_state=["Age", "Marks"], seed="observation" + ) + imputer.fit(make_df(DATA_SEED_TRAIN)) + X = make_df( + { + "City": [None, None, "Leeds"], + "Age": [None, 0.0, 30.0], + "Marks": [0.7, 0.7, 0.4], + } + ) + X_transformed = imputer.transform(X) + result = frame_to_dict(X_transformed) + assert null_count(X_transformed, "Age") == 0 + assert result["City"][0] == result["City"][1] -def test_error_if_random_state_takes_not_permitted_value(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="general", random_state="arbitrary") +def test_seed_per_observation_with_duplicated_index(): + # pandas only: polars has no index + imputer = RandomSampleImputer( + variables=["City"], random_state=["Age", "Marks"], seed="observation" + ) + imputer.fit(pd.DataFrame(DATA_SEED_TRAIN)) + X = pd.DataFrame(DATA_SEED_TEST) + expected = frame_to_dict(imputer.transform(X))["City"] -def test_error_if_random_state_is_none_when_seed_is_observation(): - with pytest.raises(ValueError): - RandomSampleImputer(seed="observation", random_state=None) + X.index = [0, 0, 1, 1, 2] + assert frame_to_dict(imputer.transform(X))["City"] == expected -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_error_if_random_state_is_string(make_df): - df_na = make_df(DATA) - with pytest.raises(ValueError): - imputer = RandomSampleImputer(seed="observation", random_state="arbitrary") - imputer.fit(df_na) +def test_error_if_random_state_variables_not_in_dataframe(make_df, data_na): + imputer = RandomSampleImputer(seed="observation", random_state="arbitrary") + msg = ( + "There are variables assigned as random state which are not part " + "of the training dataframe. Got arbitrary instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + imputer.fit(make_df(data_na)) -@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) -def test_variables_cast_as_category(make_df): - df_na = make_df(DATA) +def test_variables_cast_as_category(make_df, data_na): + df_na = make_df(data_na) if make_df is pd.DataFrame: df_na["City"] = df_na["City"].astype("category") else: @@ -280,5 +307,7 @@ def test_variables_cast_as_category(make_df): assert imputer.variables_ == ["Name", "City", "Studies", "Age", "Marks"] assert imputer.n_features_in_ == 5 - assert _null_count(X_transformed, "City") == 0 - assert set(_values(X_transformed, "City")) <= _pool(df_na, "City") + assert isinstance(X_transformed, make_df) + assert null_count(X_transformed, "City") == 0 + city_pool = {v for v in data_na["City"] if v is not None} + assert set(frame_to_dict(X_transformed)["City"]) <= city_pool