From cd7f30b2759817b3f0722bcbf8b51d436e1bfec0 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:27:15 +0200 Subject: [PATCH 1/3] Migrate the selection base classes and helpers to narwhals, add polars support BaseSelector.transform() returns the retained features in the train set order, in the same library as the input (pandas X[features], narwhals select otherwise). BaseRecursiveSelector.fit() trains the estimators on native frames and returns (nw_X, y). The helpers in base_selection_functions no longer import pandas: correlations are computed with numpy (np.corrcoef, or matrix products for pairwise complete observations when there are missing values), and feature importances are pandas Series for pandas input and dicts otherwise. Co-Authored-By: Claude Opus 5 --- .../selection/base_recursive_selector.py | 96 ++-- .../selection/base_selection_functions.py | 212 ++++++-- feature_engine/selection/base_selector.py | 44 +- tests/test_selection/conftest.py | 17 + .../test_base_recursive_selector.py | 257 +++++++++ .../test_base_selection_functions.py | 513 ++++++++++-------- tests/test_selection/test_base_selector.py | 178 ++++-- 7 files changed, 924 insertions(+), 393 deletions(-) create mode 100644 tests/test_selection/test_base_recursive_selector.py diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) From 8a105e39f563895c54bd4cf7e13a5ecc9ad25d9a Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:16:30 +0200 Subject: [PATCH 2/3] Return an empty dataframe on polars when a selector drops every feature Co-Authored-By: Claude Opus 5 --- feature_engine/selection/base_selector.py | 4 ++++ tests/test_selection/test_base_selector.py | 1 + 2 files changed, 5 insertions(+) diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index 0e4ff5d97..40e84f493 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -70,6 +70,10 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: if nwd.is_pandas_dataframe(X) is True: return X[features] else: + if len(features) == 0: + # nw.col() needs at least one name. polars frames without columns + # have no rows either. + return nw_X.select([]).to_native() return nw_X.select(nw.col(*features)).to_native() def _get_feature_names_in(self, X: IntoDataFrame): diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 7ef31226b..64e43b88c 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -74,6 +74,7 @@ def test_transform_restores_train_column_order(make_df): ([], ["Name", "City", "Age", "Marks", "dob"]), (["dob"], ["Name", "City", "Age", "Marks"]), (["Age", "Name", "City", "dob"], ["Marks"]), + (["Name", "City", "Age", "Marks", "dob"], []), ], ) def test_transform_returns_retained_features(make_df, features_to_drop, expected): From acf104fa956fcb2ae697d4c20edd8490041bc318 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:59:30 +0200 Subject: [PATCH 3/3] Migrate DropCorrelatedFeatures to narwhals, add polars support fit() validates X with check_X and hands the native dataframe to the variable, missing-value and correlation helpers, so pandas keeps its fast paths in find_correlated_features and polars is supported. The tests are rewritten to run on pandas and polars, and the user guide gets a polars example. Co-Authored-By: Claude Opus 5 --- .../selection/DropCorrelatedFeatures.rst | 56 ++- .../selection/drop_correlated_features.py | 55 ++- .../test_drop_correlated_features.py | 408 +++++++++++------- 3 files changed, 345 insertions(+), 174 deletions(-) diff --git a/docs/user_guide/selection/DropCorrelatedFeatures.rst b/docs/user_guide/selection/DropCorrelatedFeatures.rst index c83048f05..94229383a 100644 --- a/docs/user_guide/selection/DropCorrelatedFeatures.rst +++ b/docs/user_guide/selection/DropCorrelatedFeatures.rst @@ -6,12 +6,14 @@ DropCorrelatedFeatures ====================== The :class:`DropCorrelatedFeatures()` finds and removes correlated variables from a dataframe. -Correlation is calculated with `pandas.corr()`. All correlation methods supported by `pandas.corr()` -can be used in the selection, including Spearman, Kendall, or Spearman. You can also pass a -bespoke correlation function, provided it returns a value between -1 and 1. +It supports the correlation methods of `pandas.corr()`: Pearson, Spearman, and Kendall. You can +also pass a bespoke correlation function, provided it returns a value between -1 and 1. As in +`pandas.corr()`, each pair of variables is compared using the rows where both have values. + +:class:`DropCorrelatedFeatures()` works with pandas and polars dataframes. Features are removed on first found first removed basis, without any further insight. That is, -the first feature will be retained an all subsequent features that are correlated with this, will +the first feature will be retained and all subsequent features that are correlated with this, will be removed. The transformer will examine all numerical variables automatically. Note that you could pass a @@ -95,7 +97,7 @@ to `var_0` and will therefore be removed. {'var_0': {'var_8'}, 'var_4': {'var_6', 'var_7', 'var_9'}} -Similarly, `var_4` is a key and will be retained, whereas the variables 6, 7 and 8 were +Similarly, `var_4` is a key and will be retained, whereas the variables 6, 7 and 9 were found correlated to `var_4` and will therefore be removed. The features that will be removed from the dataset are stored in a different attribute @@ -137,6 +139,50 @@ Below we see the resulting dataframe: 4 -0.186530 +With polars +----------- + +:class:`DropCorrelatedFeatures()` works in the same way with a polars dataframe, and returns +a polars dataframe: + +.. code:: python + + import polars as pl + from sklearn.datasets import make_classification + from feature_engine.selection import DropCorrelatedFeatures + + X, y = make_classification(n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1) + + X = pl.DataFrame(X, schema=['var_' + str(i) for i in range(12)]) + + tr = DropCorrelatedFeatures(threshold=0.8) + Xt = tr.fit_transform(X) + + print(tr.features_to_drop_) + +The same features are found correlated as with pandas: + +.. code:: python + + ['var_8', 'var_6', 'var_7', 'var_9'] + +The transformed data is a polars dataframe without those features: + +.. code:: python + + print(Xt.columns) + +.. code:: python + + ['var_0', 'var_1', 'var_2', 'var_3', 'var_4', 'var_5', 'var_10', 'var_11'] + + Additional resources -------------------- diff --git a/feature_engine/selection/drop_correlated_features.py b/feature_engine/selection/drop_correlated_features.py index 6eefcbcda..f938b59ab 100644 --- a/feature_engine/selection/drop_correlated_features.py +++ b/feature_engine/selection/drop_correlated_features.py @@ -1,6 +1,6 @@ -from typing import List, Union +from typing import List, Optional, Union -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -47,9 +47,10 @@ ) class DropCorrelatedFeatures(BaseSelector): """ - DropCorrelatedFeatures() finds and removes correlated features. Correlation is - calculated with `pandas.corr()`. Features are removed on first found, first removed - basis, without any further insight. + DropCorrelatedFeatures() finds and removes correlated features. It supports the + correlation methods of `pandas.corr()`, and, like `pandas.corr()`, compares each + pair of features using the rows where both have values. Features are removed on + first found, first removed basis, without any further insight. DropCorrelatedFeatures() works only with numerical variables. Categorical variables will need to be encoded to numerical or will be excluded from the analysis. @@ -85,16 +86,16 @@ class DropCorrelatedFeatures(BaseSelector): Attributes ---------- features_to_drop_: - Set with the correlated features that will be dropped. + List with the correlated features that will be dropped. correlated_feature_sets_: - Groups of correlated features. Each list is a group of correlated features. + Groups of correlated features. Each set is a group of correlated features. correlated_feature_dict_: dict Dictionary containing the correlated feature groups. The key is the feature against which all other features were evaluated. The values are the features correlated with the key. Key + values should be the same as the set found in - `correlated_feature_groups`. We introduced this attribute in version 1.17.0 + `correlated_feature_sets_`. We introduced this attribute in version 1.17.0 because from the set, it is not easy to see which feature will be retained and which ones will be removed. The key is retained, the values will be dropped. @@ -139,6 +140,25 @@ class DropCorrelatedFeatures(BaseSelector): 1 2 0 2 1 0 3 1 1 + + With polars: + + >>> import polars as pl + >>> from feature_engine.selection import DropCorrelatedFeatures + >>> X = pl.DataFrame(dict(x1 = [1,2,1,1], x2 = [2,4,3,1], x3 = [1, 0, 0, 1])) + >>> dcf = DropCorrelatedFeatures(threshold=0.7) + >>> dcf.fit_transform(X) + shape: (4, 2) + ┌─────┬─────┐ + │ x1 ┆ x3 │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═════╡ + │ 1 ┆ 1 │ + │ 2 ┆ 0 │ + │ 1 ┆ 0 │ + │ 1 ┆ 1 │ + └─────┴─────┘ """ def __init__( @@ -156,7 +176,10 @@ def __init__( f"Got {threshold} instead." ) - if missing_values not in ["raise", "ignore"]: + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + ]: raise ValueError( "`missing_values` takes only values 'raise' or 'ignore'. " f"Got {missing_values} instead." @@ -169,31 +192,26 @@ def __init__( self.threshold = threshold self.missing_values = missing_values - def fit(self, X: pd.DataFrame, y: pd.Series = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Find the correlated features. Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. - y : pandas series. Default = None + y: Series, default=None y is not needed in this transformer. You can pass y or None. """ - - # check input dataframe - X = check_X(X) + check_X(X) self.variables_ = _select_numerical_variables( X, self.variables, self.confirm_variables ) - - # check that there are more than 1 variable to select from self._check_variable_number() if self.missing_values == "raise": - # check if dataset contains na _check_contains_na(X, self.variables_) _check_contains_inf(X, self.variables_) @@ -208,7 +226,6 @@ def fit(self, X: pd.DataFrame, y: pd.Series = None): self.correlated_feature_sets_ = correlated_groups self.correlated_feature_dict_ = correlated_dict - # save input features self._get_feature_names_in(X) return self diff --git a/tests/test_selection/test_drop_correlated_features.py b/tests/test_selection/test_drop_correlated_features.py index 936c2793f..c2b686f20 100644 --- a/tests/test_selection/test_drop_correlated_features.py +++ b/tests/test_selection/test_drop_correlated_features.py @@ -1,19 +1,32 @@ +import re +from datetime import datetime + import numpy as np import pandas as pd import pytest from sklearn.datasets import make_classification from feature_engine.selection import DropCorrelatedFeatures -from tests.estimator_checks.init_params_allowed_values_checks import ( - check_error_param_confirm_variables, - check_error_param_missing_values, -) +from tests.backend_helpers import frame_to_dict + +# with pearson, only a and c are correlated above 0.8. b is a monotonic, non-linear +# function of a, so the rank methods also find it; kendall doesn't find c. +DATA = { + "a": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0, 10.0], + "b": [2.7, 7.4, 20.1, 54.6, 148.4, 403.4, 1096.6, 2981.0, 8103.1, 22026.5], + "c": [2.0, 1.0, 4.0, 3.0, 6.0, 5.0, 8.0, 7.0, 10.0, 9.0], + "e": [5.0, 1.0, 9.0, 2.0, 8.0, 3.0, 10.0, 4.0, 7.0, 6.0], +} + + +def pearson(x, y): + return np.corrcoef(x, y)[0, 1] @pytest.fixture(scope="module") -def df_correlated_single(): - # create array with 4 correlated features and 2 independent ones - X, y = make_classification( +def data_correlated_single(): + """6 variables: var_1 is highly correlated with var_2 and less with var_4.""" + X, _ = make_classification( n_samples=1000, n_features=6, n_redundant=2, @@ -22,199 +35,294 @@ def df_correlated_single(): class_sep=2, random_state=1, ) + return {f"var_{i}": X[:, i].tolist() for i in range(6)} - # transform array into pandas df - colnames = ["var_" + str(i) for i in range(6)] - X = pd.DataFrame(X, columns=colnames) - return X +@pytest.fixture(scope="module") +def data_correlated_double(data_classification): + return {k: v for k, v in data_classification.items() if k != "target"} -@pytest.fixture(scope="module") -def df_correlated_double(): - # create array with 8 correlated features and 4 independent ones - X, y = make_classification( - n_samples=1000, - n_features=12, - n_redundant=4, - n_clusters_per_class=1, - weights=[0.50], - class_sep=2, - random_state=1, - ) +# init parameters +@pytest.mark.parametrize("threshold", [3, "0.1", 0, 1, 2, -0.1, 1.5, None, [0.5], True]) +def test_error_if_threshold_not_allowed(threshold): + msg = f"`threshold` must be a float between 0 and 1. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(threshold=threshold) - # transform array into pandas df - colnames = ["var_" + str(i) for i in range(12)] - X = pd.DataFrame(X, columns=colnames) - return X +@pytest.mark.parametrize("missing_values", [2, "hola", False, None, ["raise"]]) +def test_error_if_missing_values_not_allowed(missing_values): + msg = ( + "`missing_values` takes only values 'raise' or 'ignore'. " + f"Got {missing_values} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(missing_values=missing_values) -_input_params = [ - (None, "pearson", 0.8, "ignore", False), - ("var1", "kendall", 0.5, "raise", True), - (["var1", "var2"], "spearman", 0.4, "raise", False), -] +@pytest.mark.parametrize("confirm_variables", [2, "hola", [True], None]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(confirm_variables=confirm_variables) @pytest.mark.parametrize( - "_variables, _method, _threshold, _missing_values, _confirm_vars", _input_params + "method, threshold, missing_values, confirm_variables", + [ + ("pearson", 0.8, "ignore", False), + ("kendall", 0.5, "raise", True), + ("spearman", 0.0, "raise", False), + (pearson, 1.0, "ignore", True), + ], ) -def test_input_params_assignment( - _variables, _method, _threshold, _missing_values, _confirm_vars -): +def test_init_param_assignment(method, threshold, missing_values, confirm_variables): sel = DropCorrelatedFeatures( - variables=_variables, - method=_method, - threshold=_threshold, - missing_values=_missing_values, - confirm_variables=_confirm_vars, + method=method, + threshold=threshold, + missing_values=missing_values, + confirm_variables=confirm_variables, ) + assert sel.method is method + assert sel.threshold == threshold + assert sel.missing_values == missing_values + assert sel.confirm_variables is confirm_variables + + +# fit and transform +def test_default_params(make_df, data_correlated_single): + X = make_df(data_correlated_single) + sel = DropCorrelatedFeatures() + Xt = sel.fit_transform(X) + + assert sel.variables_ == ["var_0", "var_1", "var_2", "var_3", "var_4", "var_5"] + assert sel.features_to_drop_ == ["var_2"] + assert sel.correlated_feature_sets_ == [{"var_1", "var_2"}] + assert sel.correlated_feature_dict_ == {"var_1": {"var_2"}} + assert sel.feature_names_in_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + ] + assert sel.n_features_in_ == 6 + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + var: data_correlated_single[var] + for var in ["var_0", "var_1", "var_3", "var_4", "var_5"] + } - assert sel.variables == _variables - assert sel.method == _method - assert sel.threshold == _threshold - assert sel.missing_values == _missing_values - assert sel.confirm_variables == _confirm_vars - - -@pytest.mark.parametrize("_threshold", [3, "0.1", -0, 2, 0, 3, 1]) -def test_raises_error_when_threshold_not_permitted(_threshold): - msg = f"`threshold` must be a float between 0 and 1. Got {_threshold} instead." - with pytest.raises(ValueError) as record: - DropCorrelatedFeatures(threshold=_threshold) - assert record.value.args[0] == msg +def test_result_does_not_depend_on_column_order(make_df, data_correlated_single): + order = ["var_5", "var_4", "var_3", "var_2", "var_1", "var_0"] + X = make_df({var: data_correlated_single[var] for var in order}) + sel = DropCorrelatedFeatures() + Xt = sel.fit_transform(X) -def test_error_param_missing_values(): - check_error_param_missing_values(DropCorrelatedFeatures()) + assert sel.features_to_drop_ == ["var_2"] + assert sel.correlated_feature_sets_ == [{"var_1", "var_2"}] + assert sel.correlated_feature_dict_ == {"var_1": {"var_2"}} + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_5", "var_4", "var_3", "var_1", "var_0"] -def test_error_param_confirm_variables(): - check_error_param_confirm_variables(DropCorrelatedFeatures()) +def test_lower_threshold(make_df, data_correlated_single): + X = make_df(data_correlated_single) + sel = DropCorrelatedFeatures(threshold=0.6) + Xt = sel.fit_transform(X) + assert sel.features_to_drop_ == ["var_2", "var_4"] + assert sel.correlated_feature_sets_ == [{"var_1", "var_2", "var_4"}] + assert sel.correlated_feature_dict_ == {"var_1": {"var_2", "var_4"}} + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_0", "var_1", "var_3", "var_5"] -def test_default_params(df_correlated_single): - transformer = DropCorrelatedFeatures( - variables=None, method="pearson", threshold=0.8 - ) - X = transformer.fit_transform(df_correlated_single) - # expected result - df = df_correlated_single.drop("var_2", axis=1) +def test_more_than_one_correlated_group(make_df, data_correlated_double): + X = make_df(data_correlated_double) + sel = DropCorrelatedFeatures(threshold=0.6) + Xt = sel.fit_transform(X) - # test fit attrs - assert transformer.features_to_drop_ == ["var_2"] - assert transformer.correlated_feature_sets_ == [{"var_1", "var_2"}] - assert transformer.correlated_feature_dict_ == {"var_1": {"var_2"}} - # test transform output - pd.testing.assert_frame_equal(X, df) + assert sel.features_to_drop_ == ["var_8", "var_6", "var_7", "var_9"] + assert sel.correlated_feature_sets_ == [ + {"var_0", "var_8"}, + {"var_4", "var_6", "var_7", "var_9"}, + ] + assert sel.correlated_feature_dict_ == { + "var_0": {"var_8"}, + "var_4": {"var_6", "var_7", "var_9"}, + } + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + var: data_correlated_double[var] + for var in ["var_0", "var_1", "var_2", "var_3", "var_4", "var_5"] + + ["var_10", "var_11"] + } -def test_default_params_different_var_order(df_correlated_single): - transformer = DropCorrelatedFeatures( - variables=None, method="pearson", threshold=0.8 +@pytest.mark.parametrize( + "method, features_to_drop, correlated_sets, correlated_dict, retained", + [ + ("pearson", ["c"], [{"a", "c"}], {"a": {"c"}}, ["a", "b", "e"]), + ("spearman", ["b", "c"], [{"a", "b", "c"}], {"a": {"b", "c"}}, ["a", "e"]), + ("kendall", ["b"], [{"a", "b"}], {"a": {"b"}}, ["a", "c", "e"]), + (pearson, ["c"], [{"a", "c"}], {"a": {"c"}}, ["a", "b", "e"]), + ], +) +def test_correlation_methods( + make_df, method, features_to_drop, correlated_sets, correlated_dict, retained +): + sel = DropCorrelatedFeatures(method=method) + Xt = sel.fit_transform(make_df(DATA)) + + assert sel.features_to_drop_ == features_to_drop + assert sel.correlated_feature_sets_ == correlated_sets + assert sel.correlated_feature_dict_ == correlated_dict + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {var: DATA[var] for var in retained} + + +def test_negative_correlation_is_also_dropped(make_df): + X = make_df( + { + "x": [1.0, 2.0, 3.0, 4.0, 5.0, 6.0], + "y": [-2.0, -4.0, -6.0, -8.0, -10.0, -12.5], + "z": [3.0, 1.0, 2.0, 6.0, 4.0, 5.0], + } ) - var_order = list(reversed(list(df_correlated_single.columns))) - X = transformer.fit_transform(df_correlated_single[var_order]) - - # expected result - df = df_correlated_single[var_order].drop("var_2", axis=1) + sel = DropCorrelatedFeatures().fit(X) - # test fit attrs - assert transformer.features_to_drop_ == ["var_2"] - assert transformer.correlated_feature_sets_ == [{"var_1", "var_2"}] - assert transformer.correlated_feature_dict_ == {"var_1": {"var_2"}} - # test transform output - pd.testing.assert_frame_equal(X, df) + assert sel.features_to_drop_ == ["y"] + assert sel.correlated_feature_dict_ == {"x": {"y"}} -def test_lower_threshold(df_correlated_single): - transformer = DropCorrelatedFeatures( - variables=None, method="pearson", threshold=0.6 - ) - X = transformer.fit_transform(df_correlated_single) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", pearson]) +def test_missing_values_are_ignored_pairwise(make_df, method): + # y differs from 2 * x only in the row where x is missing. + data = { + "x": [1.0, 2.0, 3.0, 4.0, 5.0, None], + "y": [2.0, 4.0, 6.0, 8.0, 10.0, 1000.0], + "w": [None, 3.0, 1.0, 2.0, 5.0, 4.0], + } + sel = DropCorrelatedFeatures(method=method, missing_values="ignore") + Xt = sel.fit_transform(make_df(data)) - # expected result - df = df_correlated_single.drop(["var_2", "var_4"], axis=1) + assert sel.features_to_drop_ == ["y"] + assert sel.correlated_feature_sets_ == [{"x", "y"}] + assert sel.correlated_feature_dict_ == {"x": {"y"}} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"x": data["x"], "w": data["w"]} - # test fit attrs - assert transformer.features_to_drop_ == ["var_2", "var_4"] - assert transformer.correlated_feature_sets_ == [{"var_1", "var_2", "var_4"}] - assert transformer.correlated_feature_dict_ == {"var_1": {"var_2", "var_4"}} - # test transform output - pd.testing.assert_frame_equal(X, df) +def test_non_numerical_variables_are_ignored_and_kept(make_df): + data = { + **DATA, + "cat": ["x", "y"] * 5, + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(10)], + } + sel = DropCorrelatedFeatures() + Xt = sel.fit_transform(make_df(data)) + + assert sel.variables_ == ["a", "b", "c", "e"] + assert sel.features_to_drop_ == ["c"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + var: data[var] for var in ["a", "b", "e", "cat", "dob"] + } -def test_more_than_1_correlated_group(df_correlated_double): - transformer = DropCorrelatedFeatures( - variables=None, method="pearson", threshold=0.6 - ) - X = transformer.fit_transform(df_correlated_double) - # expected result - df = df_correlated_double.drop(["var_6", "var_7", "var_8", "var_9"], axis=1) +def test_variables_subset(make_df): + sel = DropCorrelatedFeatures(variables=["b", "c", "e"]) + Xt = sel.fit_transform(make_df(DATA)) - # test fit attrs - assert transformer.features_to_drop_ == ["var_8", "var_6", "var_7", "var_9"] - assert transformer.correlated_feature_sets_ == [ - {"var_0", "var_8"}, - {"var_4", "var_6", "var_7", "var_9"}, - ] - assert transformer.correlated_feature_dict_ == { - "var_0": {"var_8"}, - "var_4": {"var_6", "var_7", "var_9"}, - } - # test transform output - pd.testing.assert_frame_equal(X, df) + assert sel.variables_ == ["b", "c", "e"] + assert sel.features_to_drop_ == [] + assert sel.correlated_feature_sets_ == [] + assert sel.correlated_feature_dict_ == {} + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == DATA -def test_callable_method(df_correlated_double, random_uniform_method): - X = df_correlated_double +def test_confirm_variables(make_df): + sel = DropCorrelatedFeatures(variables=["a", "c", "hola"], confirm_variables=True) + Xt = sel.fit_transform(make_df(DATA)) - transformer = DropCorrelatedFeatures( - variables=None, method=random_uniform_method, threshold=0.6 - ) + assert sel.variables_ == ["a", "c"] + assert sel.features_to_drop_ == ["c"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["a", "b", "e"] - Xt = transformer.fit_transform(X) - # test no empty dataframe - assert not Xt.empty +def test_error_if_missing_values_and_raise(make_df): + X = make_df({**DATA, "a": [None] + DATA["a"][1:]}) + msg = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(missing_values="raise").fit(X) - # test fit attrs - assert len(transformer.correlated_feature_sets_) > 0 - assert len(transformer.features_to_drop_) > 0 - assert len(transformer.variables_) > 0 - assert transformer.n_features_in_ == len(X.columns) +def test_error_if_inf_and_raise(make_df): + X = make_df({**DATA, "a": [float("inf")] + DATA["a"][1:]}) + msg = ( + "Some of the variables to transform contain inf values. Check and " + "remove those before using this transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(missing_values="raise").fit(X) -def test_raises_error_when_method_not_permitted(df_correlated_double): - X = df_correlated_double - method = "hola" +def test_error_if_less_than_two_numerical_variables(make_df): + X = make_df({"a": DATA["a"], "cat": ["x", "y"] * 5}) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['a']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures().fit(X) - transformer = DropCorrelatedFeatures(variables=None, method=method, threshold=0.8) - with pytest.raises(ValueError) as errmsg: - _ = transformer.fit_transform(X) +def test_error_if_variables_not_numerical(make_df): + X = make_df({**DATA, "cat": ["x", "y"] * 5}) + msg = ( + "Some of the variables are not numerical. Please cast them as numerical " + "before using this transformer." + ) + with pytest.raises(TypeError, match=re.escape(msg)): + DropCorrelatedFeatures(variables=["a", "cat"]).fit(X) - exceptionmsg = errmsg.value.args[0] - assert ( - exceptionmsg - == "method must be either 'pearson', 'spearman', 'kendall', or a callable," - + f" '{method}' was supplied" +def test_error_if_fit_input_not_dataframe(): + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) + with pytest.raises(TypeError, match=re.escape(msg)): + DropCorrelatedFeatures().fit(np.ones((4, 3))) -def test_raises_missing_data_error(df_correlated_single): - df = df_correlated_single.copy() - df.iloc[0, 1] = np.nan +def test_error_if_method_not_allowed_pandas(): + # the message comes from pandas.DataFrame.corr(). msg = ( - "Some of the variables in the dataset contain NaN. Check and " - "remove those before using this transformer." + "method must be either 'pearson', 'spearman', 'kendall', or a callable, " + "'hola' was supplied" ) - sel = DropCorrelatedFeatures(missing_values="raise") - with pytest.raises(ValueError) as record: - sel.fit(df) - assert record.value.args[0] == msg + with pytest.raises(ValueError, match=re.escape(msg)): + DropCorrelatedFeatures(method="hola").fit(pd.DataFrame(DATA)) + + +def test_integer_column_names_pandas(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + sel = DropCorrelatedFeatures() + Xt = sel.fit_transform(X) + + assert sel.features_to_drop_ == [2] + assert sel.correlated_feature_dict_ == {0: {2}} + pd.testing.assert_frame_equal(Xt, X[[0, 1, 3]])