diff --git a/docs/user_guide/outliers/OutlierTrimmer.rst b/docs/user_guide/outliers/OutlierTrimmer.rst index 929824d45..e7ed0f0ca 100644 --- a/docs/user_guide/outliers/OutlierTrimmer.rst +++ b/docs/user_guide/outliers/OutlierTrimmer.rst @@ -125,6 +125,13 @@ and percentile methods stay closer to where the observations actually lie: they are true outliers or faithful data points. That requires further examination and domain knowledge. +.. note:: + + If all or most of the values of a variable are the same, the method may return a + spread of 0 (for example, an IQR of 0 when over half of the values are 0). The + variable then has no outliers, so :class:`OutlierTrimmer()` sets its limits to infinity + in `right_tail_caps_` and `left_tail_caps_`, and leaves it untouched. + Let’s move on to removing outliers in Python. Removing outliers in Python @@ -293,7 +300,7 @@ In the following output, we see the maximum of the variables after removing the .. code:: python fare 65.0 - age 53.0 + age 74.0 dtype: float64 Finally, we can check the boxplot of the transformed variables to corroborate the effect on their distribution. @@ -521,7 +528,7 @@ We see the adjusted data size compared to the original size here: .. code:: python - ((916, 8), (736, 76)) + ((916, 8), (828, 142)) Feature-engine's pipeline can also adjust the target: @@ -535,7 +542,7 @@ We see the adjusted data size compared to the original size here: .. code:: python - ((916,), (736,)) + ((916,), (828,)) To wrap up, let's add a machine learning algorithm to the pipeline. We'll use logistic regression to predict survival: @@ -565,7 +572,7 @@ We see the following output: .. code:: python - array([1, 1, 1, 0, 1, 0, 1, 1, 0, 1], dtype=int64) + array([1, 1, 0, 1, 0, 1, 0, 0, 1, 0]) We can obtain the probability of survival: @@ -580,16 +587,16 @@ We see the following output: .. code:: python - array([[0.13027536, 0.86972464], - [0.14982143, 0.85017857], - [0.2783799 , 0.7216201 ], - [0.86907159, 0.13092841], - [0.31794531, 0.68205469], - [0.86905145, 0.13094855], - [0.1396715 , 0.8603285 ], - [0.48403632, 0.51596368], - [0.6299007 , 0.3700993 ], - [0.49712853, 0.50287147]]) + array([[0.23320943, 0.76679057], + [0.22089305, 0.77910695], + [0.85469885, 0.14530115], + [0.28510312, 0.71489688], + [0.85468117, 0.14531883], + [0.0494853 , 0.9505147 ], + [0.58079146, 0.41920854], + [0.536129 , 0.463871 ], + [0.36885157, 0.63114843], + [0.81102131, 0.18897869]]) We can obtain the accuracy of the predictions over the test set: @@ -601,7 +608,7 @@ That returns the following accuracy: .. code:: python - 0.7823343848580442 + 0.804093567251462 We can obtain the names of the features after the transformation: @@ -635,7 +642,7 @@ We see the resulting sizes here: .. code:: python - ((393, 8), (317, 76)) + ((393, 8), (342, 142)) Setting up the stringency (param `fold`) @@ -656,6 +663,49 @@ The default values for fold are as follows: You can manually adjust the fold value to make the outlier detection process more or less conservative, thus customising the extent of outlier trimming. +With polars +----------- + +:class:`OutlierTrimmer()` works in the same way with a polars dataframe: + +.. code:: python + + import polars as pl + from feature_engine.outliers import OutlierTrimmer + + df = pl.DataFrame({ + "Age": [20, 21, 19, 18, 95], + "Marks": [0.9, 0.8, 0.7, 0.6, 0.1], + }) + + transformer = OutlierTrimmer( + capping_method="quantiles", + tail="both", + fold=0.2, + ) + + print(transformer.fit_transform(df)) + +Only the rows where both `Age` and `Marks` fall within the 20th-80th +percentile range survive; the other three rows breach the bound on at +least one of the two variables: + +.. code:: text + + shape: (2, 2) + ┌─────┬───────┐ + │ Age ┆ Marks │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═════╪═══════╡ + │ 21 ┆ 0.8 │ + │ 19 ┆ 0.7 │ + └─────┴───────┘ + +`transform_x_y()` and `get_feature_names_out()` work identically to the +pandas examples above. + + Additional resources -------------------- diff --git a/feature_engine/outliers/trimmer.py b/feature_engine/outliers/trimmer.py index 41cb48145..e86281be6 100644 --- a/feature_engine/outliers/trimmer.py +++ b/feature_engine/outliers/trimmer.py @@ -1,9 +1,10 @@ # Authors: Soledad Galli # License: BSD 3 clause -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries -from feature_engine._base_transformers.mixins import TransformXyMixin from feature_engine._docstrings.fit_attributes import ( _feature_names_in_docstring, _left_tail_caps_docstring, @@ -23,6 +24,7 @@ ) from feature_engine._docstrings.methods import _fit_transform_docstring from feature_engine._docstrings.substitute import Substitution +from feature_engine.dataframe_checks import check_X_y from feature_engine.outliers.base_outlier import WinsorizerBase @@ -41,7 +43,7 @@ n_features_in_=_n_features_in_docstring, fit_transform=_fit_transform_docstring, ) -class OutlierTrimmer(WinsorizerBase, TransformXyMixin): +class OutlierTrimmer(WinsorizerBase): """The OutlierTrimmer() removes observations with outliers from the dataset. The OutlierTrimmer() first calculates the maximum and/or minimum values @@ -174,29 +176,63 @@ class OutlierTrimmer(WinsorizerBase, TransformXyMixin): 9 0.54256 """ - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Remove observations with outliers from the dataframe. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe without outlier observations. """ + nw_X = self._check_transform_input_and_state(X) + return self._remove_outliers(nw_X).to_native() - X = self._check_transform_input_and_state(X) + def transform_x_y(self, X: IntoDataFrame, y: IntoSeries): + """ + Remove observations with outliers from the dataframe and the target. + + Parameters + ---------- + X: dataframe of shape = [n_samples, n_features] + The dataframe to transform. + + y: Series or Dataframe of length = n_samples + The target variable to transform. Can be multi-output. + + Returns + ------- + X_new: dataframe + The dataframe without outlier observations. It may contain less rows + than the original dataset. + + y_new: Series or DataFrame + The target variable, with as many rows as those left in X_new. + """ + _, y = check_X_y(X, y) + + row_index = "__row_index__" + nw_X = self._check_transform_input_and_state(X).with_row_index(row_index) + nw_X = self._remove_outliers(nw_X) + rows = nw_X.get_column(row_index).to_list() + + if nwd.is_into_series(y): + y = nw.from_native(y, series_only=True)[rows].to_native() + else: + y = nw.from_native(y, eager_only=True)[rows].to_native() + + return nw_X.drop(row_index).to_native(), y - for feature in self.right_tail_caps_.keys(): - inliers = X[feature].le(self.right_tail_caps_[feature]) - X = X.loc[inliers] + def _remove_outliers(self, nw_X: nw.DataFrame) -> nw.DataFrame: + conditions = [nw.col(f) <= c for f, c in self.right_tail_caps_.items()] + conditions += [nw.col(f) >= c for f, c in self.left_tail_caps_.items()] - for feature in self.left_tail_caps_.keys(): - inliers = X[feature].ge(self.left_tail_caps_[feature]) - X = X.loc[inliers] + if len(conditions) > 0: + nw_X = nw_X.filter(nw.all_horizontal(*conditions, ignore_nulls=False)) - return X + return nw_X diff --git a/tests/test_outliers/test_outlier_trimmer.py b/tests/test_outliers/test_outlier_trimmer.py index b4f6f8534..bc6525a12 100644 --- a/tests/test_outliers/test_outlier_trimmer.py +++ b/tests/test_outliers/test_outlier_trimmer.py @@ -2,70 +2,90 @@ # License: BSD 3 clause import numpy as np -import pandas as pd import pytest from feature_engine.outliers import OutlierTrimmer +from tests.backend_helpers import make_series, frame_to_dict +# row 0 is an outlier in both variables, row 1 only in var_b, row 4 only in var_a +DATA_TWO_VARS = {"var_a": [1, 2, 3, 4, 100], "var_b": [1000, 6, 7, 8, 9]} -def test_gaussian_right_tail_capping_when_fold_is_1(df_normal_dist): - # test case 1: mean and std, right tail + +# init parameters +# the errors come from WinsorizerBase and are tested in test_base_outlier.py +@pytest.mark.parametrize( + "capping_method, tail, fold, missing_values", + [ + ("gaussian", "right", "auto", "raise"), + ("iqr", "left", 2, "ignore"), + ("mad", "both", 1.5, "raise"), + ("quantiles", "both", 0.1, "ignore"), + ], +) +def test_init_param_assignment(capping_method, tail, fold, missing_values): + transformer = OutlierTrimmer( + capping_method=capping_method, + tail=tail, + fold=fold, + missing_values=missing_values, + ) + assert transformer.capping_method == capping_method + assert transformer.tail == tail + assert transformer.fold == fold + assert transformer.missing_values == missing_values + + +# fit and transform +def test_gaussian_right_tail_capping_when_fold_is_1(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="gaussian", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - # expected output - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].le(0.10727677848029868) - df_transf = df_transf.loc[inliers] + cap = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v <= cap] - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 83 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 83 -def test_gaussian_both_tails_capping_with_fold_2(df_normal_dist): - # test case 2: mean and std, both tails, different fold value +def test_gaussian_both_tails_capping_with_fold_2(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="gaussian", tail="both", fold=2) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - # expected output - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].between(-0.1955956473898675, 0.2075572504967645) - df_transf = df_transf.loc[inliers] + lower = transformer.left_tail_caps_["var"] + upper = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if lower <= v <= upper] - # test transform output - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 96 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 96 -def test_iqr_left_tail_capping_with_fold_2(df_normal_dist): - # test case 3: IQR, left tail, fold 2 +def test_iqr_left_tail_capping_with_fold_0_8(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="iqr", tail="left", fold=0.8) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].ge(-0.17486039103044) - df_transf = df_transf.loc[inliers] + lower = transformer.left_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v >= lower] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 98 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 98 -def test_mad_right_tail_capping_with_fold_1(df_normal_dist): - # test case 4: MAD, right tail, fold 1 +def test_mad_right_tail_capping_with_fold_1(make_df, data_normal_dist): transformer = OutlierTrimmer(capping_method="mad", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(make_df(data_normal_dist)) - df_transf = df_normal_dist.copy() - inliers = df_transf["var"].le(0.10995521088494983) - df_transf = df_transf.loc[inliers] + cap = transformer.right_tail_caps_["var"] + expected = [v for v in data_normal_dist["var"] if v <= cap] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 83 + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var": pytest.approx(expected)} + assert X.shape[0] == 83 -def test_transformer_ignores_na_in_df(df_na): - # test case 5: dataset contains na, and transformer is asked to ignore +def test_transformer_ignores_na_in_df(make_df, data_na): transformer = OutlierTrimmer( capping_method="gaussian", tail="right", @@ -73,39 +93,53 @@ def test_transformer_ignores_na_in_df(df_na): variables=["Age"], missing_values="ignore", ) - X = transformer.fit_transform(df_na) + X = transformer.fit_transform(make_df(data_na)) + + assert transformer.right_tail_caps_["Age"] == pytest.approx(38.04494616731882) + assert isinstance(X, make_df) + # rows with missing values are removed too + assert frame_to_dict(X)["Age"] == [20, 21, 19, 23, 37] - df_transf = df_na.copy() - inliers = df_transf["Age"].le(38.04494616731882) - df_transf = df_transf.loc[inliers] - pd.testing.assert_frame_equal(X, df_transf) - assert len(X) == 5 +def test_rows_are_removed_if_any_variable_is_an_outlier(make_df): + transformer = OutlierTrimmer(capping_method="quantiles", tail="both", fold=0.2) + X = transformer.fit_transform(make_df(DATA_TWO_VARS)) + assert isinstance(X, make_df) + assert frame_to_dict(X) == {"var_a": [3, 4], "var_b": [7, 8]} -def test_transform_x_t(df_normal_dist): - y = pd.Series(np.zeros(len(df_normal_dist))) + +def test_transform_x_y(make_df, data_normal_dist): + df = make_df(data_normal_dist) + y = make_series(make_df, [0.0] * len(data_normal_dist["var"])) transformer = OutlierTrimmer(capping_method="mad", tail="right", fold=1) - X = transformer.fit_transform(df_normal_dist) - assert len(X) != len(y) + X = transformer.fit_transform(df) + assert X.shape[0] != len(y) - Xt, yt = transformer.transform_x_y(df_normal_dist, y) - assert len(Xt) == len(yt) - assert len(Xt) != len(df_normal_dist) - assert (Xt.index == yt.index).all() + Xt, yt = transformer.transform_x_y(df, y) + assert isinstance(Xt, make_df) + assert isinstance(yt, type(y)) + assert frame_to_dict(Xt) == frame_to_dict(X) + assert Xt.shape[0] == len(yt) + assert Xt.shape[0] != len(data_normal_dist["var"]) @pytest.mark.parametrize( - "strings,expected", + "capping_method, expected", [("gaussian", 3), ("iqr", 1.5), ("mad", 3.29), ("quantiles", 0.05)], ) -def test_auto_fold_default_value(strings, expected, df_normal_dist): - transformer = OutlierTrimmer(capping_method=strings, fold="auto") - transformer.fit(df_normal_dist) +def test_auto_fold_default_value(capping_method, expected, make_df, data_normal_dist): + transformer = OutlierTrimmer(capping_method=capping_method, fold="auto") + transformer.fit(make_df(data_normal_dist)) assert transformer.fold_ == expected -def test_low_variation(df_normal_dist): - transformer = OutlierTrimmer(capping_method="mad") - with pytest.raises(ValueError): - transformer.fit(df_normal_dist // 10) +def test_variables_without_variation_are_left_untouched(make_df, data_normal_dist): + data = {"var": [v // 10 for v in data_normal_dist["var"]]} + transformer = OutlierTrimmer(capping_method="mad", tail="both") + Xt = transformer.fit_transform(make_df(data)) + + assert transformer.right_tail_caps_ == {"var": np.inf} + assert transformer.left_tail_caps_ == {"var": -np.inf} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == data