Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
63 changes: 63 additions & 0 deletions docs/user_guide/outliers/Winsoriser.rst
Original file line number Diff line number Diff line change
Expand Up @@ -67,6 +67,13 @@ Percentiles or quantiles
The values used by default by :class:`Winsoriser()` are those suggested as optimal
in statistical studies.

.. note::

If all or most of the values of a variable are the same, the method may return a
spread of 0 (for example, an IQR of 0 when over half of the values are 0). The
variable then has no outliers, so :class:`Winsoriser()` sets its limits to infinity
in `right_tail_caps_` and `left_tail_caps_`, and leaves it untouched.

The following image shows the four methods applied to a normal distribution. Their capping
values are close together because, when the data is roughly symmetric and bell-shaped, the
mean, median, standard deviation, IQR, and MAD all describe the same thing.
Expand Down Expand Up @@ -354,6 +361,62 @@ The default values for fold are as follows:
You can manually adjust the `fold` value to make the outlier detection process more or less
conservative, thus customising the extent of outlier capping.

With polars
-----------

:class:`Winsoriser()` works in the same way with a polars dataframe, including the
`add_indicators` option, which flags the rows that were capped on each tail:

.. code:: python

import polars as pl
from feature_engine.outliers import Winsoriser

df = pl.DataFrame({
"Age": [20, 21, 19, 18, 23, 40, 41, 97],
"Marks": [0.9, 0.8, 0.7, 0.6, 0.3, 0.5, 0.8, 0.05],
})

transformer = Winsoriser(
capping_method="iqr", tail="both", fold=1.5, add_indicators=True,
)
transformer.fit(df)

print(transformer.right_tail_caps_)
print(transformer.left_tail_caps_)

The learned capping values match those found with pandas:

.. code:: text

{'Age': 71.0, 'Marks': 1.3250000000000002}
{'Age': -11.0, 'Marks': -0.07500000000000001}

.. code:: python

print(transformer.transform(df))

`Age`'s outlier, 97, was capped to 71 and flagged in `Age_right`; none of the values
in `Marks` were extreme enough to be capped:

.. code:: text

shape: (8, 6)
┌──────┬───────┬──────────┬───────────┬────────────┬─────────────┐
│ Age ┆ Marks ┆ Age_left ┆ Age_right ┆ Marks_left ┆ Marks_right │
│ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │
│ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │
╞══════╪═══════╪══════════╪═══════════╪════════════╪═════════════╡
│ 20.0 ┆ 0.9 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 21.0 ┆ 0.8 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 19.0 ┆ 0.7 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 18.0 ┆ 0.6 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 23.0 ┆ 0.3 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 40.0 ┆ 0.5 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 41.0 ┆ 0.8 ┆ 0.0 ┆ 0.0 ┆ 0.0 ┆ 0.0 │
│ 71.0 ┆ 0.05 ┆ 0.0 ┆ 1.0 ┆ 0.0 ┆ 0.0 │
└──────┴───────┴──────────┴───────────┴────────────┴─────────────┘

Additional resources
--------------------

Expand Down
123 changes: 76 additions & 47 deletions feature_engine/outliers/winsorizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,8 +4,9 @@
import warnings
from typing import List, Literal, Union

import numpy as np
import pandas as pd
import narwhals as nw
import narwhals.dependencies as nwd
from narwhals.typing import IntoDataFrame

from feature_engine._docstrings.fit_attributes import (
_feature_names_in_docstring,
Expand All @@ -26,7 +27,6 @@
)
from feature_engine._docstrings.methods import _fit_transform_docstring
from feature_engine._docstrings.substitute import Substitution
from feature_engine.dataframe_checks import check_X
from feature_engine.outliers.base_outlier import WinsorizerBase


Expand Down Expand Up @@ -145,25 +145,33 @@ class Winsoriser(WinsorizerBase):
8 -0.469474
9 0.542560

With polars:

>>> import numpy as np
>>> import pandas as pd
>>> import polars as pl
>>> from feature_engine.outliers import Winsoriser
>>> np.random.seed(42)
>>> X = pd.DataFrame(dict(x = np.random.normal(size = 10)))
>>> X = pl.DataFrame(dict(x = np.random.normal(size = 10)))
>>> wz = Winsoriser(capping_method='mad', tail='both', fold=3)
>>> wz.fit(X)
>>> wz.transform(X)
x
0 0.496714
1 -0.138264
2 0.647689
3 1.523030
4 -0.234153
5 -0.234137
6 1.579213
7 0.767435
8 -0.469474
9 0.542560
shape: (10, 1)
┌───────────┐
│ x │
│ --- │
│ f64 │
╞═══════════╡
│ 0.496714 │
│ -0.138264 │
│ 0.647689 │
│ 1.52303 │
│ -0.234153 │
│ -0.234137 │
│ 1.579213 │
│ 0.767435 │
│ -0.469474 │
│ 0.54256 │
└───────────┘
"""

def __init__(
Expand All @@ -178,61 +186,82 @@ def __init__(
) -> None:
if not isinstance(add_indicators, bool):
raise ValueError(
"add_indicators takes only booleans True and False"
"add_indicators takes only booleans True and False. "
f"Got {add_indicators} instead."
)
super().__init__(
capping_method, tail, fold, variables, return_empty, missing_values
)
self.add_indicators = add_indicators

def transform(self, X: pd.DataFrame) -> pd.DataFrame:
def transform(self, X: IntoDataFrame) -> IntoDataFrame:
"""
Cap the variable values. Optionally, add outlier indicators.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features]
X: dataframe of shape = [n_samples, n_features]
The data to be transformed.

Returns
-------
X_new: pandas dataframe of shape = [n_samples, n_features + n_ind]
X_new: dataframe of shape = [n_samples, n_features + n_ind]
The dataframe with the capped variables and indicators.
The number of output variables depends on the values for 'tail' and
'add_indicators': if passing 'add_indicators=False', will be equal
to 'n_features', otherwise, will have an additional indicator column
per processed feature for each tail.
"""
if not self.add_indicators:
X_out = super()._transform(X)
X_out = super()._transform(X)

else:
X_orig = check_X(X)
X_out = super()._transform(X_orig)
X_orig = X_orig[self.variables_]
X_out_filtered = X_out[self.variables_]

if self.tail in ["left", "both"]:
X_left = X_out_filtered > X_orig
X_left.columns = [str(cl) + "_left" for cl in self.variables_]
if self.tail in ["right", "both"]:
X_right = X_out_filtered < X_orig
X_right.columns = [str(cl) + "_right" for cl in self.variables_]
if self.tail == "left":
X_out = pd.concat([X_out, X_left.astype(np.float64)], axis=1)
elif self.tail == "right":
X_out = pd.concat([X_out, X_right.astype(np.float64)], axis=1)
else:
X_both = pd.concat([X_left, X_right], axis=1).astype(np.float64)
X_both = X_both[
[
cl1
for cl2 in zip(X_left.columns.values, X_right.columns.values)
for cl1 in cl2
if self.add_indicators is True:
# pandas is faster than narwhals.
if nwd.is_pandas_dataframe(X_out) is True:
pd = nw.from_native(X_out, eager_only=True).__native_namespace__()
X_orig_filtered = X[self.variables_]
X_out_filtered = X_out[self.variables_]

if self.tail in ["left", "both"]:
X_left = X_out_filtered > X_orig_filtered
X_left.columns = [str(cl) + "_left" for cl in self.variables_]
if self.tail in ["right", "both"]:
X_right = X_out_filtered < X_orig_filtered
X_right.columns = [str(cl) + "_right" for cl in self.variables_]
if self.tail == "left":
X_out = pd.concat([X_out, X_left.astype("float64")], axis=1)
elif self.tail == "right":
X_out = pd.concat([X_out, X_right.astype("float64")], axis=1)
else:
X_both = pd.concat([X_left, X_right], axis=1).astype("float64")
X_both = X_both[
[
cl1
for cl2 in zip(
X_left.columns.values, X_right.columns.values
)
for cl1 in cl2
]
]
]
X_out = pd.concat([X_out, X_both], axis=1)
X_out = pd.concat([X_out, X_both], axis=1)
else:
nw_orig = nw.from_native(X, eager_only=True)
nw_out = nw.from_native(X_out, eager_only=True)

new_cols = []
for var in self.variables_:
if self.tail in ["left", "both"]:
new_cols.append(
(nw_out[var] > nw_orig[var])
.cast(nw.Float64)
.alias(f"{var}_left")
)
if self.tail in ["right", "both"]:
new_cols.append(
(nw_out[var] < nw_orig[var])
.cast(nw.Float64)
.alias(f"{var}_right")
)
X_out = nw_out.with_columns(*new_cols).to_native()

return X_out

Expand Down
Loading