From cd7f30b2759817b3f0722bcbf8b51d436e1bfec0 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:27:15 +0200 Subject: [PATCH 1/4] Migrate the selection base classes and helpers to narwhals, add polars support BaseSelector.transform() returns the retained features in the train set order, in the same library as the input (pandas X[features], narwhals select otherwise). BaseRecursiveSelector.fit() trains the estimators on native frames and returns (nw_X, y). The helpers in base_selection_functions no longer import pandas: correlations are computed with numpy (np.corrcoef, or matrix products for pairwise complete observations when there are missing values), and feature importances are pandas Series for pandas input and dicts otherwise. Co-Authored-By: Claude Opus 5 --- .../selection/base_recursive_selector.py | 96 ++-- .../selection/base_selection_functions.py | 212 ++++++-- feature_engine/selection/base_selector.py | 44 +- tests/test_selection/conftest.py | 17 + .../test_base_recursive_selector.py | 257 +++++++++ .../test_base_selection_functions.py | 513 ++++++++++-------- tests/test_selection/test_base_selector.py | 178 ++++-- 7 files changed, 924 insertions(+), 393 deletions(-) create mode 100644 tests/test_selection/test_base_recursive_selector.py diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) From 8a105e39f563895c54bd4cf7e13a5ecc9ad25d9a Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:16:30 +0200 Subject: [PATCH 2/4] Return an empty dataframe on polars when a selector drops every feature Co-Authored-By: Claude Opus 5 --- feature_engine/selection/base_selector.py | 4 ++++ tests/test_selection/test_base_selector.py | 1 + 2 files changed, 5 insertions(+) diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index 0e4ff5d97..40e84f493 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -70,6 +70,10 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: if nwd.is_pandas_dataframe(X) is True: return X[features] else: + if len(features) == 0: + # nw.col() needs at least one name. polars frames without columns + # have no rows either. + return nw_X.select([]).to_native() return nw_X.select(nw.col(*features)).to_native() def _get_feature_names_in(self, X: IntoDataFrame): diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 7ef31226b..64e43b88c 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -74,6 +74,7 @@ def test_transform_restores_train_column_order(make_df): ([], ["Name", "City", "Age", "Marks", "dob"]), (["dob"], ["Name", "City", "Age", "Marks"]), (["Age", "Name", "City", "dob"], ["Marks"]), + (["Name", "City", "Age", "Marks", "dob"], []), ], ) def test_transform_returns_retained_features(make_df, features_to_drop, expected): From b463fc12bc870676e6b36e2e5ef4737145c4dea3 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:21:00 +0200 Subject: [PATCH 3/4] Migrate RecursiveFeatureAddition to narwhals, add polars support Co-Authored-By: Claude Opus 5 --- .../selection/RecursiveFeatureAddition.rst | 99 ++- feature_engine/_docstrings/fit_attributes.py | 7 +- .../selection/recursive_feature_addition.py | 138 +-- .../test_recursive_feature_addition.py | 808 ++++++++++++------ .../test_recursive_feature_selectors.py | 21 +- 5 files changed, 717 insertions(+), 356 deletions(-) diff --git a/docs/user_guide/selection/RecursiveFeatureAddition.rst b/docs/user_guide/selection/RecursiveFeatureAddition.rst index ee1e2f4a0..95e4bf050 100644 --- a/docs/user_guide/selection/RecursiveFeatureAddition.rst +++ b/docs/user_guide/selection/RecursiveFeatureAddition.rst @@ -136,7 +136,7 @@ entire dataset: .. code:: python - 0.488702767247119 + np.float64(0.48870212980353145) Evaluating feature importance ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -202,15 +202,15 @@ In the following output we see the changes in performance returned by adding eac .. code:: python {'s1': 0, - 's5': 0.28371458794131676, - 'bmi': 0.1377714799388745, - 's2': 0.0023327265047610735, - 'bp': 0.018759914615172735, - 'sex': 0.0027996354657459643, - 's4': 0.002695149440021638, - 's3': 0.002683934134630306, - 's6': 0.000304067408860742, - 'age': -0.007387230783454768} + 's5': np.float64(0.28371458794131676), + 'bmi': np.float64(0.13777147993887456), + 's2': np.float64(0.0023327265047610735), + 'bp': np.float64(0.01875991461517268), + 'sex': np.float64(0.002799635465745798), + 's4': np.float64(0.002695149440021638), + 's3': np.float64(0.0026839341346303613), + 's6': np.float64(0.0003040674088605755), + 'age': np.float64(-0.007387230783454768)} We can also check out the standard deviation of the performance drift: @@ -225,15 +225,15 @@ returned by adding each feature: .. code:: python {'s1': 0, - 's5': 0.029336910701570382, - 'bmi': 0.01752426732750277, - 's2': 0.020525965661877265, - 'bp': 0.017326401244547558, - 'sex': 0.00867675077259389, - 's4': 0.024234566449074676, - 's3': 0.023391851139598106, - 's6': 0.016865740401721313, - 'age': 0.02042081611218045} + 's5': np.float64(0.02933691070157033), + 'bmi': np.float64(0.017524267327502716), + 's2': np.float64(0.020525965661877265), + 'bp': np.float64(0.017326401244547592), + 'sex': np.float64(0.008676750772593802), + 's4': np.float64(0.024234566449074697), + 's3': np.float64(0.02339185113959813), + 's6': np.float64(0.01686574040172137), + 'age': np.float64(0.020420816112180475)} We can now plot the performance change with the standard deviation to identify importance features: @@ -321,6 +321,67 @@ be dropped: [False, False, True, True, True, False, False, False, True, False] +With polars +~~~~~~~~~~~ + +:class:`RecursiveFeatureAddition` also selects features from polars dataframes, and +returns a polars dataframe. Let's load the diabetes dataset into a polars dataframe: + +.. code:: python + + import polars as pl + + diabetes = load_diabetes() + X = pl.DataFrame(diabetes.data, schema=diabetes.feature_names) + y = pl.Series("target", diabetes.target) + +Now, we select features as we did with pandas: + +.. code:: python + + tr = RecursiveFeatureAddition(estimator=LinearRegression(), scoring="r2", cv=3) + Xt = tr.fit_transform(X, y) + print(Xt.head()) + +The same 4 features are retained: + +.. code:: python + + shape: (5, 4) + ┌───────────┬───────────┬───────────┬───────────┐ + │ bmi ┆ bp ┆ s1 ┆ s5 │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╪═══════════╡ + │ 0.061696 ┆ 0.021872 ┆ -0.044223 ┆ 0.019907 │ + │ -0.051474 ┆ -0.026328 ┆ -0.008449 ┆ -0.068332 │ + │ 0.044451 ┆ -0.00567 ┆ -0.045599 ┆ 0.002861 │ + │ -0.011595 ┆ -0.036656 ┆ 0.012191 ┆ 0.022688 │ + │ -0.036385 ┆ 0.021872 ┆ 0.003935 ┆ -0.031988 │ + └───────────┴───────────┴───────────┴───────────┘ + +With polars, the feature importance and its standard deviation are dictionaries instead +of pandas Series: + +.. code:: python + + tr.feature_importances_ + +In the following output we see the features sorted by their importance: + +.. code:: python + + {'s1': 750.0238715216746, + 's5': 741.4713367752565, + 'bmi': 522.3301645404875, + 's2': 436.67158399913274, + 'bp': 322.0918016880965, + 'sex': 238.6195264502995, + 's4': 182.174833735295, + 's3': 113.96599187843395, + 's6': 64.76841724774016, + 'age': 41.41804062408145} + And that's it! You now know how to select features by recursively adding them to a dataset. Additional resources diff --git a/feature_engine/_docstrings/fit_attributes.py b/feature_engine/_docstrings/fit_attributes.py index 3a91a61bd..22a2b3746 100644 --- a/feature_engine/_docstrings/fit_attributes.py +++ b/feature_engine/_docstrings/fit_attributes.py @@ -35,11 +35,14 @@ # used by selection module _feature_importances_docstring = """feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. """.rstrip() _feature_importances_std_docstring = """feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. """.rstrip() _performance_drifts_docstring = """performance_drifts_: diff --git a/feature_engine/selection/recursive_feature_addition.py b/feature_engine/selection/recursive_feature_addition.py index c54185e04..a5247d23e 100644 --- a/feature_engine/selection/recursive_feature_addition.py +++ b/feature_engine/selection/recursive_feature_addition.py @@ -1,4 +1,9 @@ -import pandas as pd +from typing import Any, Dict + +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.model_selection import cross_validate from feature_engine._docstrings.fit_attributes import ( @@ -28,6 +33,7 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from feature_engine.selection.base_selection_functions import _importance_series @Substitution( @@ -144,85 +150,84 @@ class RecursiveFeatureAddition(BaseRecursiveSelector): 3 1 1 4 2 0 5 2 1 + + With polars: + + >>> import polars as pl + >>> from sklearn.ensemble import RandomForestClassifier + >>> from feature_engine.selection import RecursiveFeatureAddition + >>> X = pl.DataFrame(dict(x1 = [1000,2000,1000,1000,2000,3000], + >>> x2 = [2,4,3,1,2,2], + >>> x3 = [1,1,1,0,0,0], + >>> x4 = [1,2,1,1,0,1], + >>> x5 = [1,1,1,1,1,1])) + >>> y = pl.Series([1,0,0,1,1,0]) + >>> rfa = RecursiveFeatureAddition(RandomForestClassifier(random_state=42), cv=2) + >>> rfa.fit_transform(X, y) + shape: (6, 2) + ┌─────┬─────┐ + │ x2 ┆ x4 │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═════╡ + │ 2 ┆ 1 │ + │ 4 ┆ 2 │ + │ 3 ┆ 1 │ + │ 1 ┆ 1 │ + │ 2 ┆ 0 │ + │ 2 ┆ 1 │ + └─────┴─────┘ """ - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Find the important features. Note that the selector trains various models at each round of selection, so it might take a while. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The input dataframe + X: dataframe of shape = [n_samples, n_features] + The input dataframe. y: array-like of shape (n_samples) Target variable. Required to train the estimator. """ - X, y = super().fit(X, y) - - # Sort the feature importance values decreasingly - self.feature_importances_.sort_values(ascending=False, inplace=True) - - # Extract most important feature from the ordered list of features - first_most_important_feature = list(self.feature_importances_.index)[0] - - # Run baseline model using only the most important feature - baseline_model = cross_validate( - estimator=self.estimator, - X=X[first_most_important_feature].to_frame(), - y=y, - cv=self._cv, - groups=self.groups, - scoring=self.scoring, - return_estimator=True, + nw_X, y = super().fit(X, y) + + importances = dict(self.feature_importances_) + features = list(importances) + values = np.array(list(importances.values())) + # Same order as pandas' sort_values(ascending=False), which sorts the reversed + # values, so features with the same importance are ranked as before. + order = (len(values) - 1 - values[::-1].argsort(kind="quicksort"))[::-1] + ranked_features = [features[i] for i in order] + self.feature_importances_ = _importance_series( + X, ranked_features, values[order] ) - # Save baseline model performance - baseline_model_performance = baseline_model["test_score"].mean() + first_most_important_feature = ranked_features[0] + baseline_scores = self._cross_validate( + X, nw_X, y, [first_most_important_feature] + ) + baseline_model_performance = baseline_scores.mean() - # list to collect selected features - # It is initialized with the most important feature _selected_features = [first_most_important_feature] - - # dict to collect features and their performance_drift - # It is initialized with the performance drift of - # the most important feature - self.performance_drifts_ = {first_most_important_feature: 0} - self.performance_drifts_std_ = {first_most_important_feature: 0} - - # loop over the ordered list of features by feature importance starting - # from the second element in the list. - for feature in list(self.feature_importances_.index)[1:]: - - # Add feature and train new model - model_tmp = cross_validate( - estimator=self.estimator, - X=X[_selected_features + [feature]], - y=y, - cv=self._cv, - groups=self.groups, - scoring=self.scoring, - return_estimator=True, - ) - - # assign new model performance - model_tmp_performance = model_tmp["test_score"].mean() - - # Calculate performance drift + self.performance_drifts_: Dict[Any, float] = {first_most_important_feature: 0} + self.performance_drifts_std_: Dict[Any, float] = { + first_most_important_feature: 0 + } + + for feature in ranked_features[1:]: + scores = self._cross_validate(X, nw_X, y, _selected_features + [feature]) + model_tmp_performance = scores.mean() performance_drift = model_tmp_performance - baseline_model_performance - # Save feature and performance drift self.performance_drifts_[feature] = performance_drift - self.performance_drifts_std_[feature] = model_tmp["test_score"].std() + self.performance_drifts_std_[feature] = scores.std() - # If new performance model is if performance_drift > self.threshold: - # add feature to the list of selected features _selected_features.append(feature) - - # Update new baseline model performance baseline_model_performance = model_tmp_performance self.features_to_drop_ = [ @@ -230,3 +235,22 @@ def fit(self, X: pd.DataFrame, y: pd.Series): ] return self + + def _cross_validate( + self, X: IntoDataFrame, nw_X: nw.DataFrame, y: IntoSeries, features: list + ) -> np.ndarray: + """Return the test scores of the estimator trained on the features.""" + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + X_model = X[features] + else: + X_model = nw_X.select(nw.col(*features)).to_native() + + return cross_validate( + estimator=self.estimator, + X=X_model, + y=y, + cv=self._cv, + groups=self.groups, + scoring=self.scoring, + )["test_score"] diff --git a/tests/test_selection/test_recursive_feature_addition.py b/tests/test_selection/test_recursive_feature_addition.py index 0acd558c1..fdef405fa 100644 --- a/tests/test_selection/test_recursive_feature_addition.py +++ b/tests/test_selection/test_recursive_feature_addition.py @@ -1,310 +1,602 @@ +import re + +import numpy as np import pandas as pd import pytest -from sklearn.ensemble import RandomForestClassifier -from sklearn.linear_model import Lasso, LogisticRegression, LinearRegression -from sklearn.model_selection import KFold, GroupKFold +from sklearn.datasets import load_diabetes, make_classification +from sklearn.ensemble import HistGradientBoostingClassifier, RandomForestClassifier +from sklearn.exceptions import NotFittedError +from sklearn.linear_model import Lasso, LinearRegression, LogisticRegression +from sklearn.model_selection import GroupKFold, KFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier from sklearn.tree import DecisionTreeRegressor from feature_engine.selection import RecursiveFeatureAddition +from tests.backend_helpers import frame_to_dict, make_series -# tests for classification -_model_and_expectations = [ - ( - RandomForestClassifier(n_estimators=5, random_state=1), - 3, - 0.001, - "roc_auc", - [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", - ], - { - "var_4": 0, - "var_7": 0.0241, - "var_6": -0.001, - "var_9": -0.001, - "var_0": -0.0, - "var_8": -0.0011, - "var_10": -0.0011, - "var_11": -0.001, - "var_1": -0.0, - "var_2": -0.0001, - "var_3": -0.0011, - "var_5": -0.0001, - }, - ), - ( - LogisticRegression(random_state=10), - 2, - 0.0001, - "accuracy", - [ - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_9", - "var_10", - "var_11", - ], - { - "var_7": 0, - "var_8": 0.001, - "var_0": 0.002, - "var_6": 0.0, - "var_4": 0.0, - "var_11": -0.001, - "var_1": -0.001, - "var_5": -0.003, - "var_3": -0.002, - "var_10": 0.0, - "var_9": 0.0, - "var_2": 0.0, - }, - ), -] +@pytest.fixture(scope="module") +def data_diabetes(): + X, y = load_diabetes(return_X_y=True, as_frame=True) + data = {col: X[col].tolist() for col in X.columns} + data["target"] = y.tolist() + return data -@pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, -) -def test_classification( - estimator, cv, threshold, scoring, dropped_features, performances, df_test -): - X, y = df_test - sel = RecursiveFeatureAddition( - estimator=estimator, cv=cv, threshold=threshold, scoring=scoring - ) +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + X = rng.normal(size=(100, 5)) + data = {f"var_{i}": X[:, i].tolist() for i in range(5)} + data["target"] = (3 * X[:, 0] + X[:, 1] + rng.normal(size=100)).tolist() + groups = np.repeat(np.arange(10), 10) + rng.shuffle(groups) + return data, groups - sel.fit(X, y) - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) - # test fit attrs - assert sel.features_to_drop_ == dropped_features - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == performances +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RecursiveFeatureAddition(RandomForestClassifier(), threshold=threshold) - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +@pytest.mark.parametrize("confirm_variables", [None, 1, "True", [True]]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RecursiveFeatureAddition( + RandomForestClassifier(), confirm_variables=confirm_variables + ) -# tests for regression -_model_and_expectations = [ - ( - Lasso(alpha=0.001, random_state=10), - 3, - 0.1, - "r2", - [0, 1, 3, 4, 5, 6, 7, 9], - { - 8: 0, - 4: 0.0059, - 2: 0.1367, - 5: -0.0026, - 3: 0.0177, - 1: -0.0045, - 7: -0.0035, - 6: 0.0088, - 9: 0.002, - 0: -0.0114, - }, - ), - ( - DecisionTreeRegressor(random_state=10), - 2, - 100, - "neg_mean_squared_error", - [0, 3, 4, 5, 6, 7, 8, 9], - { - 0: -1693.8544, - 1: 106.9272, - 2: 0, - 3: -222.2945, - 4: -781.6593, - 5: -943.6299, - 6: -1701.2939, - 7: 99.315, - 8: -660.1107, - 9: -716.6378, - }, - ), -] +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (Lasso(), "neg_mean_squared_error", KFold(), [1, 2], 1, True), + (DecisionTreeRegressor(), "r2", StratifiedKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = RecursiveFeatureAddition( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + +# fit and transform @pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, + "estimator, cv, threshold, scoring, performance, drifts, dropped, retained", + [ + ( + RandomForestClassifier(n_estimators=5, random_state=1), + 3, + 0.001, + "roc_auc", + 0.9954973573949477, + { + "var_4": 0, + "var_7": 0.024094900525623353, + "var_6": -0.0010340785943196984, + "var_9": -0.0010221260221260353, + "var_0": -1.2604530676751935e-05, + "var_8": -0.001076745655058886, + "var_10": -0.0011244472840857833, + "var_11": -0.001028283407801478, + "var_1": -1.832727736339468e-05, + "var_2": -9.015137027179598e-05, + "var_3": -0.001076636995311686, + "var_5": -7.218629206573457e-05, + }, + [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_5", + "var_6", + "var_8", + "var_9", + "var_10", + "var_11", + ], + ["var_4", "var_7"], + ), + ( + LogisticRegression(random_state=10), + 2, + 0.0001, + "accuracy", + 0.99, + { + "var_7": 0, + "var_8": 0.001, + "var_0": 0.002, + "var_6": 0, + "var_4": 0, + "var_11": -0.001, + "var_1": -0.001, + "var_5": -0.003, + "var_3": -0.002, + "var_10": 0, + "var_9": 0, + "var_2": 0, + }, + [ + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_9", + "var_10", + "var_11", + ], + ["var_0", "var_7", "var_8"], + ), + ], ) -def test_regression( +def test_classification( + make_df, + data_classification, estimator, cv, threshold, scoring, - dropped_features, - performances, - load_diabetes_dataset, + performance, + drifts, + dropped, + retained, ): - # test for regression using cv=3, and the r2 as metric. - X, y = load_diabetes_dataset - - sel = RecursiveFeatureAddition( + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( estimator=estimator, cv=cv, threshold=threshold, scoring=scoring ) + Xt = selector.fit_transform(X, y) - sel.fit(X, y) + assert selector.initial_model_performance_ == pytest.approx(performance) + assert selector.performance_drifts_ == pytest.approx(drifts) + assert list(selector.performance_drifts_) == list(drifts) + assert selector.features_to_drop_ == dropped + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_classification[f] for f in retained} - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) - # test fit attrs - assert sel.features_to_drop_ == dropped_features +@pytest.mark.parametrize( + "estimator, cv, threshold, scoring, drifts, dropped, retained", + [ + ( + Lasso(alpha=0.001, random_state=10), + 3, + 0.1, + "r2", + { + "s5": 0, + "s1": 0.005876977602335132, + "bmi": 0.1367220291124882, + "s2": -0.002593999790115431, + "bp": 0.017686140335452738, + "sex": -0.004478777508460652, + "s4": -0.003491685019777313, + "s3": 0.008817074774180478, + "s6": 0.002042365258727974, + "age": -0.011362436906615925, + }, + ["age", "sex", "bp", "s1", "s2", "s3", "s4", "s6"], + ["bmi", "s5"], + ), + ( + DecisionTreeRegressor(random_state=10), + 2, + 100, + "neg_mean_squared_error", + { + "bmi": 0, + "s5": -660.1106667923586, + "s2": -943.6298975615891, + "s4": 99.31498680241202, + "bp": -222.29449032177035, + "s6": -716.6378161136254, + "s3": -1701.2939247109098, + "age": -1693.8544450729005, + "s1": -781.6593093262954, + "sex": 106.92720965309127, + }, + ["age", "bp", "s1", "s2", "s3", "s4", "s5", "s6"], + ["sex", "bmi"], + ), + ], +) +def test_regression( + make_df, data_diabetes, estimator, cv, threshold, scoring, drifts, dropped, retained +): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition( + estimator=estimator, cv=cv, threshold=threshold, scoring=scoring + ) + Xt = selector.fit_transform(X, y) - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == performances - - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) - - -def test_performance_drift_std(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3) - sel.fit(X, y) - - drifts = { - 4: 0, - 8: 0.2837, - 2: 0.1378, - 5: 0.0023, - 3: 0.0188, - 1: 0.0028, - 7: 0.0027, - 6: 0.0027, - 9: 0.0003, - 0: -0.0074, - } + assert selector.performance_drifts_ == pytest.approx(drifts) + assert list(selector.performance_drifts_) == list(drifts) + assert selector.features_to_drop_ == dropped + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_diabetes[f] for f in retained} - drfts_std = { - 4: 0, - 8: 0.0293, - 2: 0.0175, - 5: 0.0205, - 3: 0.0173, - 1: 0.0087, - 7: 0.0242, - 6: 0.0234, - 9: 0.0169, - 0: 0.0204, - } - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == drifts +def test_attributes_with_linear_regression(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) - rounded_perfs = { - key: round(sel.performance_drifts_std_[key], 4) - for key in sel.performance_drifts_std_ - } - assert rounded_perfs == drfts_std - - -def test_feature_importance(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3) - sel.fit(X, y) - - imps = [ - 750.02, - 741.47, - 522.33, - 436.67, - 322.09, - 238.62, - 182.17, - 113.97, - 64.77, - 41.42, + assert selector.initial_model_performance_ == pytest.approx(0.48870212980353145) + # the importance is sorted from the most to the least important feature. + assert dict(selector.feature_importances_) == pytest.approx( + { + "s1": 750.0238715216746, + "s5": 741.4713367752565, + "bmi": 522.3301645404875, + "s2": 436.67158399913274, + "bp": 322.0918016880965, + "sex": 238.6195264502995, + "s4": 182.174833735295, + "s3": 113.96599187843395, + "s6": 64.76841724774016, + "age": 41.41804062408145, + } + ) + assert list(selector.feature_importances_.keys()) == [ + "s1", + "s5", + "bmi", + "s2", + "bp", + "sex", + "s4", + "s3", + "s6", + "age", ] - imps_std = [18.22, 68.35, 86.03, 57.11, 329.38, 299.76, 72.81, 47.93, 117.83, 42.75] + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "age": 18.21715207624677, + "sex": 68.35471931174253, + "bmi": 86.03069812358049, + "bp": 57.11038282196886, + "s1": 329.3758185655728, + "s2": 299.7569984100255, + "s3": 72.80549636690168, + "s4": 47.925822175574034, + "s5": 117.8299487852095, + "s6": 42.75477362271488, + } + ) + assert selector.performance_drifts_ == pytest.approx( + { + "s1": 0, + "s5": 0.28371458794131676, + "bmi": 0.13777147993887456, + "s2": 0.0023327265047611845, + "bp": 0.018759914615172624, + "sex": 0.0027996354657458533, + "s4": 0.0026951494400216935, + "s3": 0.002683934134630417, + "s6": 0.00030406740886079753, + "age": -0.007387230783454712, + } + ) + assert selector.performance_drifts_std_ == pytest.approx( + { + "s1": 0, + "s5": 0.02933691070157033, + "bmi": 0.017524267327502716, + "s2": 0.020525965661877258, + "bp": 0.01732640124454753, + "sex": 0.008676750772593741, + "s4": 0.024234566449074697, + "s3": 0.02339185113959813, + "s6": 0.016865740401721358, + "age": 0.020420816112180495, + } + ) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] - assert round(sel.feature_importances_, 2).to_list() == imps - assert round(sel.feature_importances_std_, 2).to_list() == imps_std +def test_feature_importances_type(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) -def test_cv_generator(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - cv = KFold(n_splits=3) - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3).fit(X, y) - expected = sel.transform(X) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=cv).fit( - X, y + +def test_ranking_of_features_with_equal_importance(make_df): + # Lasso sets most coefficients to 0. With more than 16 features the sort is not + # stable, so both backends must rank these ties as pandas' sort_values does. + X, y = make_classification( + n_samples=500, n_features=25, n_informative=4, random_state=3 + ) + X = make_df({f"var_{i}": X[:, i] for i in range(25)}) + selector = RecursiveFeatureAddition(Lasso(alpha=0.05), scoring="r2", cv=3) + selector.fit(X, make_series(make_df, y)) + + assert list(selector.feature_importances_.keys()) == [ + "var_22", + "var_13", + "var_16", + "var_9", + "var_19", + "var_11", + "var_0", + "var_14", + "var_23", + "var_21", + "var_20", + "var_18", + "var_17", + "var_15", + "var_12", + "var_1", + "var_10", + "var_8", + "var_7", + "var_6", + "var_5", + "var_4", + "var_3", + "var_2", + "var_24", + ] + assert list(selector.performance_drifts_) == list( + selector.feature_importances_.keys() ) - test1 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test1) - sel = RecursiveFeatureAddition( - estimator=linear_model, scoring="r2", cv=cv.split(X, y) - ).fit(X, y) - test2 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test2) +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so the importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( + KNeighborsClassifier(), scoring="accuracy", cv=3 + ) + Xt = selector.fit_transform(X, y) -def test_recursive_feature_addition_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) - threshold = 0.5 + assert list(selector.feature_importances_.keys())[:3] == [ + "var_7", + "var_6", + "var_4", + ] + assert selector.feature_importances_["var_7"] == pytest.approx(0.2313333333333333) + assert selector.features_to_drop_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_8", + "var_9", + "var_10", + "var_11", + ] + assert frame_to_dict(Xt) == {"var_7": data_classification["var_7"]} - estimator = LinearRegression() - scoring = "neg_mean_absolute_error" - sel_expected = RecursiveFeatureAddition( - estimator=estimator, - scoring=scoring, - cv=cv_indices, - threshold=threshold, +def test_variables_and_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( + LogisticRegression(), + cv=3, + variables=["var_0", "var_4", "var_7", "var_9", "Hola"], + confirm_variables=True, ) + Xt = selector.fit_transform(X, y) - X_tr_expected = sel_expected.fit_transform(X, y) + assert selector.variables_ == ["var_0", "var_4", "var_7", "var_9"] + assert selector.performance_drifts_ == pytest.approx( + { + "var_7": 0, + "var_0": 0.0007903910012343474, + "var_4": 0.0004547772620060453, + "var_9": 0.0005748100627617214, + } + ) + assert selector.features_to_drop_ == ["var_0", "var_4", "var_9"] + assert list(Xt.columns) == [ + "var_1", + "var_2", + "var_3", + "var_5", + "var_6", + "var_7", + "var_8", + "var_10", + "var_11", + ] - sel = RecursiveFeatureAddition( - estimator=estimator, - scoring=scoring, - cv=cv, + +def test_non_numerical_variables_are_ignored_and_kept(make_df, data_classification): + data = { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "var_7": data_classification["var_7"], + "target": data_classification["target"], + } + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition(LogisticRegression(), cv=3) + Xt = selector.fit_transform(X, y) + + assert selector.variables_ == ["var_0", "var_4", "var_7"] + assert list(selector.performance_drifts_) == ["var_7", "var_0", "var_4"] + assert selector.features_to_drop_ == ["var_0", "var_4"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"cat": data["cat"], "var_7": data["var_7"]} + + +def test_fit_with_missing_values(make_df, data_classification): + # The selector doesn't check for NaN: estimators that handle it can be used. + data = dict(data_classification) + for var in ["var_0", "var_4", "var_7"]: + data[var] = [None if i % 20 == 0 else v for i, v in enumerate(data[var])] + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition( + HistGradientBoostingClassifier(max_iter=10, random_state=0), + scoring="accuracy", + cv=3, + ) + Xt = selector.fit_transform(X, y) + + assert selector.features_to_drop_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_8", + "var_9", + "var_10", + "var_11", + ] + assert frame_to_dict(Xt) == {f: data[f] for f in ["var_6", "var_7"]} + + +def test_transform_returns_features_in_training_order(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + Xt = selector.transform(X[list(X.columns)[::-1]]) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_diabetes[f] for f in ["bmi", "bp", "s1", "s5"]} + + +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_cv_splitter_and_generator(make_df, data_diabetes, cv_type): + X, y = _split_target(make_df, data_diabetes) + cv = KFold(n_splits=3) + if cv_type == "generator": + cv = cv.split(X, y) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=cv) + selector.fit(X, y) + + assert selector.performance_drifts_["s5"] == pytest.approx(0.28371458794131676) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] + + +def test_cv_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition( + LinearRegression(), + scoring="r2", + cv=GroupKFold(n_splits=3), groups=groups, - threshold=threshold, + threshold=0.05, + ) + selector.fit(X, y) + + assert selector.performance_drifts_ == pytest.approx( + { + "var_0": 0, + "var_1": 0.07658061969570773, + "var_4": 0.0015218369914395957, + "var_2": -0.009235024059350394, + "var_3": -0.002472987677324623, + } + ) + assert selector.features_to_drop_ == ["var_2", "var_3", "var_4"] + + indices = GroupKFold(n_splits=3).split(X, y, groups) + selector_indices = RecursiveFeatureAddition( + LinearRegression(), scoring="r2", cv=indices, threshold=0.05 ) - X_tr = sel.fit_transform(X, y) + selector_indices.fit(X, y) - pd.testing.assert_frame_equal( - X_tr_expected, - X_tr, + assert selector_indices.performance_drifts_ == pytest.approx( + selector.performance_drifts_ ) + assert selector_indices.features_to_drop_ == ["var_2", "var_3", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_list_and_array_target(make_df, data_diabetes, target_type): + X, _ = _split_target(make_df, data_diabetes) + y = target_type(data_diabetes["target"]) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + + assert selector.performance_drifts_["s5"] == pytest.approx(0.28371458794131676) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] + + +def test_error_if_transform_before_fit(make_df, data_diabetes): + X, _ = _split_target(make_df, data_diabetes) + msg = ( + "This RecursiveFeatureAddition instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + RecursiveFeatureAddition(LinearRegression()).transform(X) + + +def test_feature_importances_are_pandas_series(data_diabetes): + X, y = _split_target(pd.DataFrame, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [ + 750.0238715216746, + 741.4713367752565, + 522.3301645404875, + 436.67158399913274, + 322.0918016880965, + 238.6195264502995, + 182.174833735295, + 113.96599187843395, + 64.76841724774016, + 41.41804062408145, + ], + index=["s1", "s5", "bmi", "s2", "bp", "sex", "s4", "s3", "s6", "age"], + ), + ) + + +def test_integer_column_names(data_diabetes): + X, y = _split_target(pd.DataFrame, data_diabetes) + X.columns = list(range(10)) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + Xt = selector.fit_transform(X, y) + + assert selector.features_to_drop_ == [0, 1, 5, 6, 7, 9] + assert list(selector.performance_drifts_) == [4, 8, 2, 5, 3, 1, 7, 6, 9, 0] + pd.testing.assert_frame_equal(Xt, X[[2, 3, 4, 8]]) diff --git a/tests/test_selection/test_recursive_feature_selectors.py b/tests/test_selection/test_recursive_feature_selectors.py index d9d3aebc8..239bb2591 100644 --- a/tests/test_selection/test_recursive_feature_selectors.py +++ b/tests/test_selection/test_recursive_feature_selectors.py @@ -10,14 +10,10 @@ from sklearn.svm import SVR from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor -from feature_engine.selection import ( - RecursiveFeatureAddition, - RecursiveFeatureElimination, -) +from feature_engine.selection import RecursiveFeatureElimination _selectors = [ RecursiveFeatureElimination, - RecursiveFeatureAddition, ] _input_params = [ @@ -158,12 +154,6 @@ def test_fit_initial_model_performance( def test_feature_importances(_estimator, _importance, df_test): X, y = df_test - sel = RecursiveFeatureAddition(_estimator, threshold=-100).fit(X, y) - _importance.sort(reverse=True) - assert ( - list(np.round(sel.feature_importances_.values, 2)) == np.round(_importance, 2) - ).all() - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) _importance.sort(reverse=False) assert ( @@ -201,11 +191,6 @@ def test_permutation_importance(_estimator, df_test): expected_importances.index = X.columns expected_importances = expected_importances.mean(axis=1) - sel = RecursiveFeatureAddition(_estimator, threshold=-100).fit(X, y) - assert_series_equal( - sel.feature_importances_.sort_index(), expected_importances.sort_index() - ) - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) assert_series_equal( sel.feature_importances_.sort_index(), expected_importances.sort_index() @@ -215,10 +200,6 @@ def test_permutation_importance(_estimator, df_test): @pytest.mark.parametrize("_estimator", ests) def test_selection_after_permutation_importance(_estimator, df_test): X, y = df_test - sel = RecursiveFeatureAddition(_estimator) - Xtr = sel.fit_transform(X, y) - assert Xtr.shape[1] < X.shape[1] - sel = RecursiveFeatureElimination(_estimator) Xtr = sel.fit_transform(X, y) assert Xtr.shape[1] < X.shape[1] From dca28a53d736f31fb08a92db2d9e1bddf759cd20 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:50:09 +0200 Subject: [PATCH 4/4] Mark polars output blocks in the user guide as text Co-Authored-By: Claude Opus 5 --- docs/user_guide/selection/RecursiveFeatureAddition.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/user_guide/selection/RecursiveFeatureAddition.rst b/docs/user_guide/selection/RecursiveFeatureAddition.rst index 95e4bf050..90287033c 100644 --- a/docs/user_guide/selection/RecursiveFeatureAddition.rst +++ b/docs/user_guide/selection/RecursiveFeatureAddition.rst @@ -345,7 +345,7 @@ Now, we select features as we did with pandas: The same 4 features are retained: -.. code:: python +.. code:: text shape: (5, 4) ┌───────────┬───────────┬───────────┬───────────┐