From cd7f30b2759817b3f0722bcbf8b51d436e1bfec0 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:27:15 +0200 Subject: [PATCH 1/3] Migrate the selection base classes and helpers to narwhals, add polars support BaseSelector.transform() returns the retained features in the train set order, in the same library as the input (pandas X[features], narwhals select otherwise). BaseRecursiveSelector.fit() trains the estimators on native frames and returns (nw_X, y). The helpers in base_selection_functions no longer import pandas: correlations are computed with numpy (np.corrcoef, or matrix products for pairwise complete observations when there are missing values), and feature importances are pandas Series for pandas input and dicts otherwise. Co-Authored-By: Claude Opus 5 --- .../selection/base_recursive_selector.py | 96 ++-- .../selection/base_selection_functions.py | 212 ++++++-- feature_engine/selection/base_selector.py | 44 +- tests/test_selection/conftest.py | 17 + .../test_base_recursive_selector.py | 257 +++++++++ .../test_base_selection_functions.py | 513 ++++++++++-------- tests/test_selection/test_base_selector.py | 178 ++++-- 7 files changed, 924 insertions(+), 393 deletions(-) create mode 100644 tests/test_selection/test_base_recursive_selector.py diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) From 08df647a7627d045caca51e53d82a0a3a815bb0c Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:25:59 +0200 Subject: [PATCH 2/3] Migrate SelectByShuffling to narwhals, add polars support SelectByShuffling works with pandas, polars and other narwhals-supported dataframes. For a given random_state the shuffles, drifts and selected features are the same as before and the same with every library. The validation folds come from cross_validate(return_indices=True), so each model is scored on the fold it was not trained on, also when the splitter gives different folds on every call. The fold dataframes are built once and one column is shuffled at a time, instead of copying the whole dataframe for every feature. Co-Authored-By: Claude Opus 5 --- .../selection/SelectByShuffling.rst | 89 +++- feature_engine/selection/shuffle_features.py | 149 +++--- tests/test_selection/test_shuffle_features.py | 452 +++++++++++++----- 3 files changed, 486 insertions(+), 204 deletions(-) diff --git a/docs/user_guide/selection/SelectByShuffling.rst b/docs/user_guide/selection/SelectByShuffling.rst index 61043945d..7b5654c93 100644 --- a/docs/user_guide/selection/SelectByShuffling.rst +++ b/docs/user_guide/selection/SelectByShuffling.rst @@ -104,7 +104,7 @@ the entire dataset, without shuffling, using cross-validation. .. code:: python - 0.488702767247119 + 0.48870212980353145 In the following sections, we'll explore some of the additional useful data stored by :class:`SelectByShuffling()`. @@ -125,15 +125,15 @@ each feature: .. code:: python {'age': -0.0054698043007869734, - 'sex': 0.03325633986510784, - 'bmi': 0.184158237207512, + 'sex': 0.03325633986510779, + 'bmi': 0.1841582372075119, 'bp': 0.10089894421748086, - 's1': 0.49324432634948095, - 's2': 0.21163252880660438, - 's3': 0.02006839198785859, - 's4': 0.011098050006761673, - 's5': 0.4828781996541602, - 's6': 0.003963360084439538} + 's1': 0.4932443263494815, + 's2': 0.2116325288066046, + 's3': 0.020068391987858536, + 's4': 0.011098050006761617, + 's5': 0.4828781996541603, + 's6': 0.003963360084439482} :class:`SelectByShuffling()` stores the standard deviation of the performance change: @@ -146,16 +146,16 @@ shuffling: .. code:: python - {'age': 0.012788500580799392, + {'age': 0.012788500580799462, 'sex': 0.040792331972680645, - 'bmi': 0.042212436355346106, - 'bp': 0.05397012536801143, - 's1': 0.35198797776358015, - 's2': 0.167636042355086, - 's3': 0.03455158514716544, - 's4': 0.007755675852874145, - 's5': 0.1449579162698361, - 's6': 0.011193022434166025} + 'bmi': 0.04221243635534631, + 'bp': 0.053970125368011726, + 's1': 0.3519879777635807, + 's2': 0.1676360423550863, + 's3': 0.03455158514716549, + 's4': 0.007755675852874141, + 's5': 0.14495791626983628, + 's6': 0.011193022434166125} We can plot the performance change together with the standard deviation to get a better idea of how shuffling features affect the model performance: @@ -181,8 +181,8 @@ feature: .. figure:: ../../images/shuffle-features-std.png -With this set up, features that elicited a mean performance drop greater than the mean -performance of all features, will be removed. If, for any reason, this threshold is too +With this set up, features that elicited a performance drop smaller than the mean +performance drop of all features will be removed. If, for any reason, this threshold is too conservative or too permissive, by analysing the former barplot, you can get a better idea of how these features affect the predictions of the model, and select a different threshold. @@ -198,7 +198,7 @@ threshold: tr.features_to_drop_ The following features were deemed as non-important, because their performance drift is -greater than the mean performance drift of all features: +smaller than the mean performance drift of all features: .. code:: python @@ -221,7 +221,52 @@ In the following output, we see the dataframe with the selected features: 3 -0.011595 0.012191 0.024991 0.022688 4 -0.036385 0.003935 0.015596 -0.031988 - +Using polars +~~~~~~~~~~~~ + +:class:`SelectByShuffling()` also works with polars dataframes, and returns a polars +dataframe when the input is a polars dataframe. With the same `random_state`, the +features are shuffled in the same way as with pandas, so the performance drifts and the +selected features are the same. + +.. code:: python + + import polars as pl + + X_pl = pl.DataFrame(X.to_dict(orient="list")) + y_pl = pl.Series("target", y.to_list()) + + tr = SelectByShuffling( + estimator=LinearRegression(), + scoring="r2", + cv=3, + random_state=0, + ) + + Xt = tr.fit_transform(X_pl, y_pl) + + print(tr.features_to_drop_) + print(Xt.head()) + +In the following output, we see the features to drop and the polars dataframe with the +selected features: + +.. code:: python + + ['age', 'sex', 'bp', 's3', 's4', 's6'] + shape: (5, 4) + ┌───────────┬───────────┬───────────┬───────────┐ + │ bmi ┆ s1 ┆ s2 ┆ s5 │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╪═══════════╡ + │ 0.061696 ┆ -0.044223 ┆ -0.034821 ┆ 0.019907 │ + │ -0.051474 ┆ -0.008449 ┆ -0.019163 ┆ -0.068332 │ + │ 0.044451 ┆ -0.045599 ┆ -0.034194 ┆ 0.002861 │ + │ -0.011595 ┆ 0.012191 ┆ 0.024991 ┆ 0.022688 │ + │ -0.036385 ┆ 0.003935 ┆ 0.015596 ┆ -0.031988 │ + └───────────┴───────────┴───────────┴───────────┘ + Additional resources -------------------- diff --git a/feature_engine/selection/shuffle_features.py b/feature_engine/selection/shuffle_features.py index 946b963d3..a3d66273b 100644 --- a/feature_engine/selection/shuffle_features.py +++ b/feature_engine/selection/shuffle_features.py @@ -1,11 +1,13 @@ from types import GeneratorType from typing import List, MutableSequence, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd -from sklearn.base import is_classifier +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.metrics import get_scorer -from sklearn.model_selection import check_cv, cross_validate +from sklearn.model_selection import cross_validate +from sklearn.utils import _safe_indexing from sklearn.utils.validation import _check_sample_weight, check_random_state from feature_engine._check_init_parameters.check_variables import ( @@ -169,6 +171,37 @@ class SelectByShuffling(BaseSelector): 3 1 1 1 4 2 0 1 5 2 1 1 + + With polars: + + >>> import polars as pl + >>> from sklearn.ensemble import RandomForestClassifier + >>> from feature_engine.selection import SelectByShuffling + >>> X = pl.DataFrame(dict(x1 = [1000,2000,1000,1000,2000,3000], + >>> x2 = [2,4,3,1,2,2], + >>> x3 = [1,1,1,0,0,0], + >>> x4 = [1,2,1,1,0,1], + >>> x5 = [1,1,1,1,1,1])) + >>> y = pl.Series([1,0,0,1,1,0]) + >>> sbs = SelectByShuffling( + >>> RandomForestClassifier(random_state=42), + >>> cv=2, + >>> random_state=42, + >>> ) + >>> sbs.fit_transform(X, y) + shape: (6, 3) + ┌─────┬─────┬─────┐ + │ x2 ┆ x4 ┆ x5 │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ i64 │ + ╞═════╪═════╪═════╡ + │ 2 ┆ 1 ┆ 1 │ + │ 4 ┆ 2 ┆ 1 │ + │ 3 ┆ 1 ┆ 1 │ + │ 1 ┆ 1 ┆ 1 │ + │ 2 ┆ 0 ┆ 1 │ + │ 2 ┆ 1 ┆ 1 │ + └─────┴─────┴─────┘ """ def __init__( @@ -182,8 +215,11 @@ def __init__( confirm_variables: bool = False, ): - if threshold and not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float or None") + if threshold is not None and not isinstance(threshold, (int, float)): + raise ValueError( + "threshold must be an integer, a float or None. " + f"Got {threshold} instead." + ) super().__init__(confirm_variables) @@ -196,8 +232,8 @@ def __init__( def fit( self, - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y: IntoSeries, sample_weight: Union[MutableSequence, None] = None, ): """ @@ -205,8 +241,9 @@ def fit( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The input dataframe. + X: dataframe of shape = [n_samples, n_features] + The input dataframe. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: array-like of shape (n_samples) Target variable. Required to train the estimator. @@ -214,12 +251,7 @@ def fit( sample_weight : array-like of shape (n_samples,), default=None Sample weights. If None, then samples are equally weighted. """ - - X, y = check_X_y(X, y) - - # reset the index - X = X.reset_index(drop=True) - y = y.reset_index(drop=True) + nw_X, y = check_X_y(X, y) if sample_weight is not None: sample_weight = _check_sample_weight(sample_weight, X) @@ -233,70 +265,76 @@ def fit( cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv + nw_X_model = nw_X.select(nw.col(*self.variables_)) + X_model = nw_X_model.to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=cv, return_estimator=True, + return_indices=True, scoring=self.scoring, params={"sample_weight": sample_weight}, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # extract the validation folds - cv_ = check_cv(cv, y=y, classifier=is_classifier(self.estimator)) - validation_indices = [val_index for _, val_index in cv_.split(X, y)] + # the indices returned by cross_validate are the folds each model was + # evaluated on, also when the splitter gives different folds on every call. + validation_indices = model["indices"]["test"] + y_val = [_safe_indexing(y, idx) for idx in validation_indices] - # get performance metric scorer = get_scorer(self.scoring) - - # seed random_state = check_random_state(self.random_state) + n_samples = nw_X.shape[0] - # dict to collect features and their performance_drift after shuffling self.performance_drifts_ = {} self.performance_drifts_std_ = {} - # shuffle features and save feature performance drift into a dict - for feature in self.variables_: - - X_shuffled = X[self.variables_].copy() + estimators = model["estimator"] - # shuffle individual feature - X_shuffled[feature] = ( - X_shuffled[feature] - .sample(frac=1, random_state=random_state) - .reset_index(drop=True) - ) - - # determine the performance with the shuffled feature - performance = [ - scorer(m, X_shuffled.iloc[idx], y.iloc[idx]) - for m, idx in zip(model["estimator"], validation_indices) - ] - - performance_std = np.std(performance) - performance = np.mean(performance) - - # determine drift in performance - # Note, sklearn negates the log and error scores, so no need to manually - # do the inversion - # https://scikit-learn.org/stable/modules/model_evaluation.html - # (https://scikit-learn.org/stable/modules/model_evaluation.html - # #the-scoring-parameter-defining-model-evaluation-rules) - performance_drift = self.initial_model_performance_ - performance + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + X_val = [X_model.iloc[idx] for idx in validation_indices] + else: + nw_X_val = [nw_X_model[idx] for idx in validation_indices] - # Save feature and performance drift - self.performance_drifts_[feature] = performance_drift - self.performance_drifts_std_[feature] = performance_std + for feature in self.variables_: + # same permutation as pandas' sample(frac=1), so the drifts for a given + # random_state are the same with every dataframe library. + permutation = random_state.permutation(n_samples) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + values = X_model[feature].array + performance = [] + for m, idx, X_, y_ in zip(estimators, validation_indices, X_val, y_val): + # the fold dataframes are copies made in fit, so we can shuffle + # the column in place and restore it afterwards. + original = X_[feature] + X_[feature] = values.take(permutation[idx]) + performance.append(scorer(m, X_, y_)) + X_[feature] = original + else: + column = nw_X_model.get_column(feature) + performance = [ + scorer(m, X_.with_columns(column[permutation[idx]]).to_native(), y_) + for m, idx, X_, y_ in zip( + estimators, validation_indices, nw_X_val, y_val + ) + ] + + # sklearn negates the error and loss scores, so larger is always better. + drift = self.initial_model_performance_ - np.mean(performance) + self.performance_drifts_[feature] = drift + self.performance_drifts_std_[feature] = np.std(performance) # select features if not self.threshold: - threshold = pd.Series(self.performance_drifts_).mean() + threshold = np.mean(list(self.performance_drifts_.values())) else: threshold = self.threshold @@ -306,7 +344,6 @@ def fit( if self.performance_drifts_[f] < threshold ] - # save input features self._get_feature_names_in(X) return self diff --git a/tests/test_selection/test_shuffle_features.py b/tests/test_selection/test_shuffle_features.py index 5f1e65dbb..badfbdaa9 100644 --- a/tests/test_selection/test_shuffle_features.py +++ b/tests/test_selection/test_shuffle_features.py @@ -1,192 +1,392 @@ +import re + import numpy as np import pandas as pd import pytest - -from sklearn.ensemble import RandomForestClassifier -from sklearn.linear_model import LinearRegression -from sklearn.model_selection import StratifiedKFold +from sklearn.datasets import load_diabetes +from sklearn.ensemble import HistGradientBoostingClassifier, RandomForestClassifier +from sklearn.linear_model import LinearRegression, LogisticRegression +from sklearn.model_selection import KFold, StratifiedKFold from sklearn.tree import DecisionTreeRegressor from feature_engine.selection import SelectByShuffling +from tests.backend_helpers import frame_to_dict, make_series + +VARIABLES = ["var_0", "var_4", "var_7", "var_9"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _random_forest(): + return RandomForestClassifier(n_estimators=10, random_state=1) + + +@pytest.fixture(scope="module") +def data_diabetes(): + X, y = load_diabetes(return_X_y=True, as_frame=True) + data = {col: X[col].tolist() for col in X.columns} + data["target"] = y.tolist() + return data -def test_sel_with_default_parameters(df_test): - X, y = df_test +# init parameters +@pytest.mark.parametrize("threshold", ["hello", "", [0.1], [], {"a": 1}, (1,)]) +def test_error_if_threshold_not_number_or_none(threshold): + msg = f"threshold must be an integer, a float or None. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + SelectByShuffling(_random_forest(), threshold=threshold) + + +@pytest.mark.parametrize("confirm_variables", ["True", 1, None, [True]]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + SelectByShuffling(_random_forest(), confirm_variables=confirm_variables) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, threshold, random_state, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, None, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), 0.01, 1, True), + (LinearRegression(), "r2", KFold(2), 5, 10, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, threshold, random_state, confirm_variables +): sel = SelectByShuffling( - RandomForestClassifier(random_state=1), threshold=0.01, random_state=1 + estimator, + scoring=scoring, + cv=cv, + threshold=threshold, + random_state=random_state, + confirm_variables=confirm_variables, + ) + assert sel.estimator is estimator + assert sel.scoring == scoring + assert sel.cv is cv + assert sel.threshold == threshold + assert sel.random_state == random_state + assert sel.confirm_variables is confirm_variables + + +# fit and transform +def test_fit_attributes(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 ) sel.fit(X, y) - # expected result - Xtransformed = pd.DataFrame(X["var_7"].copy()) + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.0037026895460631204, + "var_4": 0.0017710452198405058, + "var_7": 0.25884577270119447, + "var_9": -1.959497441428315e-05, + } + ) + assert sel.performance_drifts_std_ == pytest.approx( + { + "var_0": 0.0005724303014393721, + "var_4": 0.0005148668118348698, + "var_7": 0.02708631411106102, + "var_9": 0.0016362909674612345, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + assert sel.variables_ == VARIABLES + assert sel.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert sel.n_features_in_ == 12 + + +def test_transform_removes_features_below_threshold(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = SelectByShuffling(_random_forest(), threshold=0.01, random_state=1) + Xt = sel.fit_transform(X, y) - # test init params - assert sel.threshold == 0.01 - assert sel.cv == 3 - assert sel.scoring == "roc_auc" - # test fit attrs - assert np.round(sel.initial_model_performance_, 3) == 0.997 + assert sel.initial_model_performance_ == pytest.approx(0.9953593232960704) assert sel.features_to_drop_ == [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", + f"var_{i}" for i in [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11] ] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_7": data_classification["var_7"]} -def test_regression_cv_3_and_r2(load_diabetes_dataset): - # test for regression using cv=3, and the r2 as metric. - X, y = load_diabetes_dataset - +def test_threshold_none_uses_mean_drift(make_df, data_classification): + X, y = _split_target(make_df, data_classification) sel = SelectByShuffling( - estimator=LinearRegression(), scoring="r2", cv=3, threshold=0.05, random_state=1 + LogisticRegression(), scoring="accuracy", variables=VARIABLES, random_state=3 ) sel.fit(X, y) - # expected output - Xtransformed = X[[1, 2, 3, 4, 5, 8]].copy() + # the mean drift is 0.13, only var_7 is above it. + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.021998045950141765, + "var_4": 0.005002007996020019, + "var_7": 0.4890009770249292, + "var_9": 0.004001006995019041, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] - # test init params - assert sel.cv == 3 - assert sel.scoring == "r2" - assert sel.threshold == 0.05 - # fit params - assert np.round(sel.initial_model_performance_, 3) == 0.489 - assert sel.features_to_drop_ == [0, 6, 7, 9] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +def test_regression_with_r2(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + sel = SelectByShuffling( + LinearRegression(), scoring="r2", cv=3, threshold=0.05, random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.48870212980353145) + assert sel.performance_drifts_ == pytest.approx( + { + "age": 0.0019221800216577822, + "sex": 0.058082587710752975, + "bmi": 0.16770566452186242, + "bp": 0.0702676565845552, + "s1": 0.5151999913097192, + "s2": 0.17198756874519755, + "s3": 0.01669668794929896, + "s4": 0.025518182049075577, + "s5": 0.4065911011104548, + "s6": 0.0038028879119369474, + } + ) + assert sel.features_to_drop_ == ["age", "s3", "s4", "s6"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["sex", "bmi", "bp", "s1", "s2", "s5"] -def test_regression_cv_2_and_mse(load_diabetes_dataset): - # test for regression using cv=2, and the neg_mean_squared_error as metric. - # add suitable threshold for regression mse - X, y = load_diabetes_dataset +def test_regression_with_neg_mse_and_target_as_dataframe(make_df, data_diabetes): + X, _ = _split_target(make_df, data_diabetes) + y = make_df({"target": data_diabetes["target"]}) sel = SelectByShuffling( - estimator=DecisionTreeRegressor(random_state=0), + DecisionTreeRegressor(random_state=0), scoring="neg_mean_squared_error", cv=2, threshold=1000, random_state=1, ) - # fit transformer sel.fit(X, y) - # expected output - Xtransformed = X[[2, 7, 8]].copy() + assert sel.initial_model_performance_ == pytest.approx(-5835.58371040724) + assert sel.performance_drifts_ == pytest.approx( + { + "age": -227.27149321266916, + "sex": -78.94570135746653, + "bmi": 2132.739819004524, + "bp": 134.5814479638011, + "s1": 279.30316742081413, + "s2": 313.13800904977325, + "s3": 19.719457013574356, + "s4": 2050.5294117647063, + "s5": 1661.9977375565613, + "s6": -4.617647058823422, + } + ) + assert sel.features_to_drop_ == ["age", "sex", "bp", "s1", "s2", "s3", "s6"] + + +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_cv_options(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, cv=cv, threshold=0.01, random_state=1 + ) + sel.fit(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + - # test init params - assert sel.cv == 2 - assert sel.scoring == "neg_mean_squared_error" - assert sel.threshold == 1000 - # fit params - assert np.round(sel.initial_model_performance_, 0) == -5836.0 - assert sel.features_to_drop_ == [0, 1, 3, 4, 5, 6, 9] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +class _FoldsChangeOnEachCall(KFold): + """Splitter that yields different folds every time split() is called.""" + def split(self, X, y=None, groups=None): + self.n_calls = getattr(self, "n_calls", 0) + 1 + folds = list(super().split(X, y, groups)) + return iter(folds if self.n_calls % 2 == 1 else folds[::-1]) -def test_cv_generator(df_test): - X, y = df_test - cv = StratifiedKFold(n_splits=3) - X, y = df_test +def test_performance_is_evaluated_on_the_folds_of_each_model( + make_df, data_classification +): + # the shuffled data must be scored on the fold each model was not trained on. + X, y = _split_target(make_df, data_classification) + params = dict(variables=VARIABLES, random_state=1) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, - random_state=1, - cv=3, + _random_forest(), cv=_FoldsChangeOnEachCall(n_splits=3), **params ) sel.fit(X, y) + reference = SelectByShuffling(_random_forest(), cv=KFold(n_splits=3), **params) + reference.fit(X, y) + + assert sel.performance_drifts_ == pytest.approx(reference.performance_drifts_) + assert sel.performance_drifts_std_ == pytest.approx( + reference.performance_drifts_std_ + ) + + +def test_non_numerical_variables_are_ignored(make_df, data_classification): + data = {**data_classification, "cat_1": ["a"] * 1000, "cat_2": ["b"] * 1000} + X, y = _split_target(make_df, data) + sel = SelectByShuffling(_random_forest(), threshold=0.01, random_state=1) + Xt = sel.fit_transform(X, y) - # expected result - Xtransformed = pd.DataFrame(X["var_7"].copy()) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert sel.variables_ == [f"var_{i}" for i in range(12)] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_7": data_classification["var_7"], + "cat_1": ["a"] * 1000, + "cat_2": ["b"] * 1000, + } + +def test_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, + _random_forest(), + variables=["var_0", "var_4", "var_7", "not_in_X"], + confirm_variables=True, random_state=1, - cv=cv, ) sel.fit(X, y) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert sel.variables_ == ["var_0", "var_4", "var_7"] + assert sel.features_to_drop_ == ["var_0", "var_4"] + + +def test_missing_values(make_df, data_classification): + data = {k: data_classification[k] for k in [*VARIABLES, "target"]} + data["var_0"] = [None if i % 7 == 0 else v for i, v in enumerate(data["var_0"])] + data["var_7"] = [None if i % 5 == 0 else v for i, v in enumerate(data["var_7"])] + X, y = _split_target(make_df, data) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, - random_state=1, - cv=cv.split(X, y), + HistGradientBoostingClassifier(max_iter=10, random_state=0), random_state=1 ) sel.fit(X, y) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + + assert sel.initial_model_performance_ == pytest.approx(0.993134587411696) + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.008290557612846916, + "var_4": 0.35213096252252896, + "var_7": 0.002123827199128625, + "var_9": 0.000764131562324577, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_7", "var_9"] -def test_raises_threshold_error(): - with pytest.raises(ValueError): - SelectByShuffling(RandomForestClassifier(random_state=1), threshold="hello") +def test_sample_weight(make_df): + X = make_df( + { + "x1": [1000, 2000, 1000, 1000, 2000, 3000], + "x2": [1000, 2000, 1000, 1000, 2000, 3000], + } + ) + y = make_series(make_df, [1, 0, 0, 1, 1, 0]) + sel = SelectByShuffling( + RandomForestClassifier(random_state=42), cv=2, random_state=42 + ) + sel.fit(X, y, sample_weight=[1000, 2000, 1000, 1000, 2000, 3000]) + assert sel.initial_model_performance_ == 0.125 + assert sel.features_to_drop_ == ["x2"] -def test_automatic_variable_selection(df_test): - X, y = df_test - # add 2 additional categorical variables, these should not be evaluated by - # the selector - X["cat_1"] = "cat1" - X["cat_2"] = "cat2" +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_and_array(make_df, data_classification, to_target): + X, _ = _split_target(make_df, data_classification) + y = to_target(data_classification["target"]) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), threshold=0.01, random_state=1 + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 ) sel.fit(X, y) - # expected result - Xtransformed = X[["var_7", "cat_1", "cat_2"]].copy() + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] - # test init params - assert sel.threshold == 0.01 - assert sel.cv == 3 - assert sel.scoring == "roc_auc" - # test fit attrs - assert np.round(sel.initial_model_performance_, 3) == 0.997 - assert sel.features_to_drop_ == [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", - ] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +def test_input_dataframe_is_not_modified(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + SelectByShuffling(_random_forest(), variables=VARIABLES, random_state=1).fit(X, y) -def test_sample_weights(): - X = pd.DataFrame( - dict( - x1=[1000, 2000, 1000, 1000, 2000, 3000], - x2=[1000, 2000, 1000, 1000, 2000, 3000], - ) + assert frame_to_dict(X) == { + k: v for k, v in data_classification.items() if k != "target" + } + + +def test_integer_column_names_pandas(data_classification): + X = pd.DataFrame({i: data_classification[f"var_{i}"] for i in range(12)}) + y = pd.Series(data_classification["target"]) + sel = SelectByShuffling( + _random_forest(), variables=[0, 4, 7, 9], threshold=0.01, random_state=1 ) - y = pd.Series([1, 0, 0, 1, 1, 0]) + Xt = sel.fit_transform(X, y) + + assert sel.performance_drifts_ == pytest.approx( + { + 0: 0.0037026895460631204, + 4: 0.0017710452198405058, + 7: 0.25884577270119447, + 9: -1.959497441428315e-05, + } + ) + assert sel.features_to_drop_ == [0, 4, 9] + pd.testing.assert_frame_equal(Xt, X.drop(columns=[0, 4, 9])) - sbs = SelectByShuffling( - RandomForestClassifier(random_state=42), cv=2, random_state=42 + +def test_non_default_index_pandas(data_classification): + index = np.arange(1000)[::-1] + 50 + X = pd.DataFrame( + {k: v for k, v in data_classification.items() if k != "target"}, index=index ) + y = pd.Series(data_classification["target"], index=index) + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + pd.testing.assert_frame_equal(Xt, X.drop(columns=["var_0", "var_4", "var_9"])) + - sample_weight = [1000, 2000, 1000, 1000, 2000, 3000] - sbs.fit_transform(X, y, sample_weight=sample_weight) - assert sbs.initial_model_performance_ == 0.125 +def test_nullable_integer_dtype_pandas(data_classification): + X = pd.DataFrame({k: data_classification[k] for k in VARIABLES}) + X["var_0"] = pd.array( + [None if i % 7 == 0 else round(v * 10) for i, v in enumerate(X["var_0"])], + dtype="Int64", + ) + y = pd.Series(data_classification["target"]) + sel = SelectByShuffling( + HistGradientBoostingClassifier(max_iter=10, random_state=0), random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.variables_ == VARIABLES + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.0007341414720931638, + "var_4": -0.00020804719599920585, + "var_7": 0.45245947715827217, + "var_9": 0.0001250311491274303, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + pd.testing.assert_frame_equal(Xt, X[["var_7"]]) From b030c09b4fc0057a90f66a49b801cde0cacda537 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:50:11 +0200 Subject: [PATCH 3/3] Mark polars output blocks in the user guide as text Co-Authored-By: Claude Opus 5 --- docs/user_guide/selection/SelectByShuffling.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/user_guide/selection/SelectByShuffling.rst b/docs/user_guide/selection/SelectByShuffling.rst index 7b5654c93..b144c274d 100644 --- a/docs/user_guide/selection/SelectByShuffling.rst +++ b/docs/user_guide/selection/SelectByShuffling.rst @@ -251,7 +251,7 @@ selected features are the same. In the following output, we see the features to drop and the polars dataframe with the selected features: -.. code:: python +.. code:: text ['age', 'sex', 'bp', 's3', 's4', 's6'] shape: (5, 4)