Skip to content
32 changes: 13 additions & 19 deletions docs/user_guide/imputation/RandomSampleImputer.rst
Original file line number Diff line number Diff line change
Expand Up @@ -28,12 +28,12 @@ missing data and `seed` is the number you entered in the `random_state`.

If `seed = 'observation'`, then the random_state should be a variable name
or a list of variable names. The seed will be calculated observation per
observation, either by adding or multiplying the values of the variables
indicated in the `random_state`. Then, a value will be extracted from the train set
using that seed and used to replace the NAN in that particular observation. This is the
equivalent of `pandas.sample(1, random_state=var1+var2)` if the `seeding_method` is
set to `add` or `pandas.sample(1, random_state=var1*var2)` if the `seeding_method`
is set to `multiply`.
observation from the values of the variables indicated in the `random_state`.
Then, a value will be extracted from the train set using that seed and used to
replace the NAN in that particular observation.

Observations with the same values in the `random_state` variables receive the same
imputation, regardless of their position in the dataframe.

For example, if the observation shows variables colour: np.nan, height: 152, weight:52,
and we set the imputer as:
Expand All @@ -43,20 +43,16 @@ and we set the imputer as:
RandomSampleImputer(
random_state=['height', 'weight'],
seed='observation',
seeding_method='add',
)

the np.nan in the variable colour will be replaced using pandas sample as follows:

.. code:: python

observation.sample(1, random_state=int(152+52))
the np.nan in the variable colour will be replaced with a value extracted from the train
set, using a seed derived from the values 152 and 52. Any other observation with
height 152 and weight 52 will receive the same value.

.. note::

Note, if the variables indicated in the `random_state` list are not numerical
the imputer will return an error. In addition, the variables indicated as seed
should not contain missing values themselves.
The variables indicated in the `random_state` must be numerical, otherwise the
imputer will return an error. Missing values in those variables are treated as 0.

With polars
-----------
Expand All @@ -79,7 +75,6 @@ With polars
variables=["LotFrontage"],
random_state=["MSSubClass", "YrSold"],
seed="observation",
seeding_method="add",
)
imputer.fit(X_train)
imputer.transform(X_train)
Expand Down Expand Up @@ -146,8 +141,8 @@ First, let's load the data and separate it into train and test:
)

In this example, we sample values at random, observation per observation, using as seed
the value of the variable 'MSSubClass' plus the value of the variable 'YrSold'. Note
that the seed's value is different for each observation.
the values of the variables 'MSSubClass' and 'YrSold'. Observations with the same values
in these variables receive the same imputed values.

The :class:`RandomSampleImputer()` will impute all variables in the data, as we left the
default value of the parameter `variables` to `None`.
Expand All @@ -158,7 +153,6 @@ default value of the parameter `variables` to `None`.
imputer = RandomSampleImputer(
random_state=['MSSubClass', 'YrSold'],
seed='observation',
seeding_method='add'
)

# fit the imputer
Expand Down
5 changes: 4 additions & 1 deletion feature_engine/imputation/arbitrary_imputer.py
Original file line number Diff line number Diff line change
Expand Up @@ -154,7 +154,10 @@ def __init__(
if isinstance(arbitrary_number, int) or isinstance(arbitrary_number, float):
self.arbitrary_number = arbitrary_number
else:
raise ValueError("arbitrary_number must be numeric of type int or float")
raise ValueError(
"arbitrary_number must be numeric of type int or float. "
f"Got {arbitrary_number} instead."
)

_check_numerical_dict(imputer_dict)

Expand Down
19 changes: 16 additions & 3 deletions feature_engine/imputation/categorical.py
Original file line number Diff line number Diff line change
Expand Up @@ -148,13 +148,26 @@ def __init__(
return_object: bool = False,
ignore_format: bool = False,
) -> None:
if imputation_method not in ["missing", "frequent"]:
if not isinstance(imputation_method, str) or imputation_method not in [
"missing",
"frequent",
]:
raise ValueError(
"imputation_method takes only values 'missing' or 'frequent'"
"imputation_method takes only values 'missing' or 'frequent'. "
f"Got {imputation_method} instead."
)

if not isinstance(ignore_format, bool):
raise ValueError("ignore_format takes only booleans True and False")
raise ValueError(
"ignore_format takes only booleans True and False. "
f"Got {ignore_format} instead."
)

if not isinstance(return_object, bool):
raise ValueError(
"return_object takes only booleans True and False. "
f"Got {return_object} instead."
)

self.imputation_method = imputation_method
self.fill_value = fill_value
Expand Down
10 changes: 5 additions & 5 deletions feature_engine/imputation/drop_missing_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -256,15 +256,15 @@ def _select_rows(self, X: IntoDataFrame, keep: bool) -> IntoDataFrame:
# dropna(subset=[]) keeps every row: there are no variables to
# evaluate missingness on, so nothing can ever be "missing".
if keep is True:
return X
if nwd.is_pandas_dataframe(X):
return X.iloc[:0]
return X.to_native()
return X.head(0).to_native()

# Benchmarked: a numpy-backed mask beats both pandas' own axis=1
# isnull()/notna().sum() (a known-slow reduction) and the narwhals
# path below, so pandas keeps this dedicated fast path.
if nwd.is_pandas_dataframe(X):
# path below, so pandas keeps this dedicated fast path. X is the
# narwhals frame returned by check_X, so branch on its implementation.
if X.implementation.is_pandas():
X = X.to_native()
if self.threshold is not None:
non_null_count = X[self.variables_].notna().to_numpy().sum(axis=1)
mask = non_null_count >= len(self.variables_) * self.threshold
Expand Down
19 changes: 13 additions & 6 deletions feature_engine/imputation/end_tail.py
Original file line number Diff line number Diff line change
Expand Up @@ -173,16 +173,23 @@ def __init__(
return_empty: bool = False,
) -> None:

if imputation_method not in ["gaussian", "iqr", "max"]:
if not isinstance(imputation_method, str) or imputation_method not in [
"gaussian",
"iqr",
"max",
]:
raise ValueError(
"imputation_method takes only values 'gaussian', 'iqr' or 'max'"
"imputation_method takes only values 'gaussian', 'iqr' or 'max'. "
f"Got {imputation_method} instead."
)

if tail not in ["right", "left"]:
raise ValueError("tail takes only values 'right' or 'left'")
if not isinstance(tail, str) or tail not in ["right", "left"]:
raise ValueError(
f"tail takes only values 'right' or 'left'. Got {tail} instead."
)

if fold <= 0:
raise ValueError("fold takes only positive numbers")
if not isinstance(fold, (int, float)) or isinstance(fold, bool) or fold <= 0:
raise ValueError(f"fold takes only positive numbers. Got {fold} instead.")

self.imputation_method = imputation_method
self.tail = tail
Expand Down
10 changes: 8 additions & 2 deletions feature_engine/imputation/mean_median.py
Original file line number Diff line number Diff line change
Expand Up @@ -136,8 +136,14 @@ def __init__(
return_empty: bool = False,
) -> None:

if imputation_method not in ["median", "mean"]:
raise ValueError("imputation_method takes only values 'median' or 'mean'")
if not isinstance(imputation_method, str) or imputation_method not in [
"median",
"mean",
]:
raise ValueError(
"imputation_method takes only values 'median' or 'mean'. "
f"Got {imputation_method} instead."
)

self.imputation_method = imputation_method
self.variables = _check_variables_input_value(variables)
Expand Down
5 changes: 4 additions & 1 deletion feature_engine/imputation/missing_indicator.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,7 +142,10 @@ def __init__(
) -> None:

if not isinstance(missing_only, bool):
raise ValueError("missing_only takes values True or False")
raise ValueError(
"missing_only takes values True or False. "
f"Got {missing_only} instead."
)

self.variables = _check_variables_input_value(variables)
self.missing_only = missing_only
Expand Down
112 changes: 52 additions & 60 deletions feature_engine/imputation/random_sample.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
# Authors: Soledad Galli <solegalli@protonmail.com>
# License: BSD 3 clause

import hashlib
from typing import List, Optional, Union

import narwhals.dependencies as nwd
Expand Down Expand Up @@ -32,21 +33,27 @@
from feature_engine.variable_handling import check_all_variables, find_all_variables


# for RandomSampleImputer
def _define_seed(
X: IntoDataFrame,
index: int,
seed_variables: Union[str, int, List[Union[str, int]]],
how: str = "add",
) -> int:
# Pandas-only: relies on .loc label-based row access, so it is only
# called from the pandas branch of transform(), where X is already
# confirmed to be a pandas dataframe.
if how == "add":
internal_seed = int(np.round(X.loc[index, seed_variables].sum(), 0))
elif how == "multiply":
internal_seed = int(np.round(X.loc[index, seed_variables].product(), 0))
return internal_seed
def _hash_seeds(values) -> np.ndarray:
"""Return one seed per row, in [0, 2**32), derived from the row's values.

Rows with the same values get the same seed, regardless of their position.
Values are compared as floats (25 and 25.0 are equal) and missing values
count as 0. hashlib, unlike hash(), gives the same seed in every session.
"""
values = np.asarray(values, dtype="float64")
values = values.reshape(len(values), -1)
# + 0.0 turns -0.0 into 0.0, so both give the same bytes
values = np.where(np.isnan(values), 0.0, values) + 0.0
values = np.ascontiguousarray(values, dtype="<f8")
return np.array(
[
int.from_bytes(
hashlib.blake2b(row.tobytes(), digest_size=4).digest(), "little"
)
for row in values
],
dtype=np.int64,
)


@Substitution(
Expand Down Expand Up @@ -93,14 +100,11 @@ class RandomSampleImputer(BaseImputer):
**'general'**: one seed will be used to impute the entire dataframe. This is
equivalent to setting the seed in pandas.sample(random_state).

**'observation'**: the seed will be set for each observation using the values
**'observation'**: the seed will be set for each observation from the values
of the variables indicated in the random_state for that particular
observation.

seeding_method: str, default='add'
If more than one variable is indicated to seed the random sampling per
observation, you can choose to combine those values as an addition or a
multiplication. Can take the values 'add' or 'multiply'.
observation. Observations with the same values in those variables receive
the same imputation, regardless of their position in the dataframe.
Missing values in those variables are treated as 0.

Attributes
----------
Expand Down Expand Up @@ -172,25 +176,26 @@ def __init__(
return_empty: bool = False,
random_state: Union[None, int, str, List[Union[str, int]]] = None,
seed: str = "general",
seeding_method: str = "add",
) -> None:

if seed not in ["general", "observation"]:
raise ValueError("seed takes only values 'general' or 'observation'")

if seeding_method not in ["add", "multiply"]:
raise ValueError("seeding_method takes only values 'add' or 'multiply'")
if not isinstance(seed, str) or seed not in ["general", "observation"]:
raise ValueError(
"seed takes only values 'general' or 'observation'. "
f"Got {seed} instead."
)

if seed == "general" and random_state:
if not isinstance(random_state, int):
raise ValueError(
"if seed == 'general' then random_state must take an integer"
"if seed == 'general' then random_state must take an integer. "
f"Got {random_state} instead."
)

if seed == "observation" and not random_state:
raise ValueError(
"if seed == 'observation' the random state must take the name of one "
"or more variables which will be used to seed the imputer"
"or more variables which will be used to seed the imputer. "
f"Got {random_state} instead."
)

self.variables = _check_variables_input_value(variables)
Expand All @@ -200,7 +205,6 @@ def __init__(

self.random_state = random_state
self.seed = seed
self.seeding_method = seeding_method

def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None):
"""
Expand Down Expand Up @@ -244,7 +248,7 @@ def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None):
):
raise ValueError(
"There are variables assigned as random state which are not part "
"of the training dataframe."
f"of the training dataframe. Got {self.random_state} instead."
)
self.random_state = random_state

Expand Down Expand Up @@ -308,26 +312,21 @@ def _transform_pandas(self, X):

# random sampling observation per observation
elif self.seed == "observation" and self.random_state:
# seeds come from the values before any variable is imputed; rows are
# addressed by position, so duplicated index labels don't matter
seeds = _hash_seeds(X[self.random_state].to_numpy())
for feature in self.variables_:
if X[feature].isnull().sum() > 0:

# loop over each observation with missing data
for i in X[X[feature].isnull()].index:
# find the seed using additional variables
internal_seed = _define_seed(
X, i, self.random_state, how=self.seeding_method
)

# extract 1 value at random
random_sample = (
self.X_[feature]
.dropna()
.sample(1, replace=True, random_state=internal_seed)
)
random_sample = random_sample.values[0]

# replace the missing data point
X.loc[i, feature] = random_sample
is_null = X[feature].isnull().to_numpy()
if is_null.any():
pool = self.X_[feature].dropna()
positions = np.flatnonzero(is_null)
random_values = [
pool.sample(
1, replace=True, random_state=int(seeds[pos])
).iloc[0]
for pos in positions
]
X.iloc[positions, X.columns.get_loc(feature)] = random_values
return X

def _transform_narwhals(self, X):
Expand All @@ -351,15 +350,8 @@ def _transform_narwhals(self, X):
X = X.with_columns(col.scatter(positions, random_sample))

elif self.seed == "observation" and self.random_state:
# Vectorized stand-in for pandas' .loc-based per-row seed lookup:
# narwhals dataframes are positional (no row labels), so the seed
# for every row is computed up-front with numpy instead of in a
# per-row .loc lookup.
seed_values = X.select(self.random_state).to_numpy()
if self.seeding_method == "add":
internal_seeds = np.round(seed_values.sum(axis=1), 0).astype(int)
else:
internal_seeds = np.round(seed_values.prod(axis=1), 0).astype(int)
# seeds come from the values before any variable is imputed
internal_seeds = _hash_seeds(X.select(self.random_state).to_numpy())

for feature in self.variables_:
col = X[feature]
Expand Down
Loading