From cd7f30b2759817b3f0722bcbf8b51d436e1bfec0 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 11:27:15 +0200 Subject: [PATCH 1/2] Migrate the selection base classes and helpers to narwhals, add polars support BaseSelector.transform() returns the retained features in the train set order, in the same library as the input (pandas X[features], narwhals select otherwise). BaseRecursiveSelector.fit() trains the estimators on native frames and returns (nw_X, y). The helpers in base_selection_functions no longer import pandas: correlations are computed with numpy (np.corrcoef, or matrix products for pairwise complete observations when there are missing values), and feature importances are pandas Series for pandas input and dicts otherwise. Co-Authored-By: Claude Opus 5 --- .../selection/base_recursive_selector.py | 96 ++-- .../selection/base_selection_functions.py | 212 ++++++-- feature_engine/selection/base_selector.py | 44 +- tests/test_selection/conftest.py | 17 + .../test_base_recursive_selector.py | 257 +++++++++ .../test_base_selection_functions.py | 513 ++++++++++-------- tests/test_selection/test_base_selector.py | 178 ++++-- 7 files changed, 924 insertions(+), 393 deletions(-) create mode 100644 tests/test_selection/test_base_recursive_selector.py diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) From 0cf549e4073f3c215217ac99b9e0bdb51ae937b9 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sat, 19 Sep 2026 12:15:00 +0200 Subject: [PATCH 2/2] Migrate RecursiveFeatureElimination to narwhals, add polars support fit() uses the (nw_X, y) returned by BaseRecursiveSelector.fit and trains each round's models on native frames (pandas X[features], narwhals select otherwise). feature_importances_ is sorted with pandas' sort_values for pandas input and with numpy's argsort (the same quicksort) for dicts, so tied features are evaluated in the same order on both backends. Tests rewritten to the make_df conventions; the RFE cases in test_recursive_feature_selectors.py moved to the RFE test file. The user guide gets a polars example and a corrected initial model performance. Co-Authored-By: Claude Opus 5 --- .../selection/RecursiveFeatureElimination.rst | 63 ++- .../recursive_feature_elimination.py | 79 ++- .../test_recursive_feature_elimination.py | 499 ++++++++++-------- .../test_recursive_feature_selectors.py | 21 +- 4 files changed, 410 insertions(+), 252 deletions(-) diff --git a/docs/user_guide/selection/RecursiveFeatureElimination.rst b/docs/user_guide/selection/RecursiveFeatureElimination.rst index 709936693..552e2984a 100644 --- a/docs/user_guide/selection/RecursiveFeatureElimination.rst +++ b/docs/user_guide/selection/RecursiveFeatureElimination.rst @@ -144,7 +144,7 @@ entire dataset: .. code:: python - 0.488702767247119 + 0.48870212980353145 Evaluating feature importance ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -257,7 +257,7 @@ features: r['mean'].plot.bar(yerr=[r['std'], r['std']], subplots=True) - plt.title("Performance drift elicited by adding features") + plt.title("Performance drift elicited by removing features") plt.ylabel('Mean performance drift') plt.xlabel('Features') plt.show() @@ -327,6 +327,65 @@ be dropped: [False, True, True, True, True, True, False, False, True, False] +With polars +~~~~~~~~~~~ + +:class:`RecursiveFeatureElimination` also accepts polars dataframes. The selection is +the same as with pandas, and `transform()` returns a polars dataframe: + +.. code:: python + + import polars as pl + from sklearn.datasets import load_diabetes + from sklearn.linear_model import LinearRegression + from feature_engine.selection import RecursiveFeatureElimination + + data = load_diabetes() + X = pl.DataFrame(data.data, schema=data.feature_names) + y = pl.Series("target", data.target) + + tr = RecursiveFeatureElimination(estimator=LinearRegression(), scoring="r2", cv=3) + Xt = tr.fit_transform(X, y) + print(Xt.head()) + +In the following output we see the same six features that we selected from the pandas +dataframe: + +.. code:: text + + shape: (5, 6) + ┌───────────┬───────────┬───────────┬───────────┬───────────┬───────────┐ + │ sex ┆ bmi ┆ bp ┆ s1 ┆ s2 ┆ s5 │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╪═══════════╪═══════════╪═══════════╡ + │ 0.05068 ┆ 0.061696 ┆ 0.021872 ┆ -0.044223 ┆ -0.034821 ┆ 0.019907 │ + │ -0.044642 ┆ -0.051474 ┆ -0.026328 ┆ -0.008449 ┆ -0.019163 ┆ -0.068332 │ + │ 0.05068 ┆ 0.044451 ┆ -0.00567 ┆ -0.045599 ┆ -0.034194 ┆ 0.002861 │ + │ -0.044642 ┆ -0.011595 ┆ -0.036656 ┆ 0.012191 ┆ 0.024991 ┆ 0.022688 │ + │ -0.044642 ┆ -0.036385 ┆ 0.021872 ┆ 0.003935 ┆ 0.015596 ┆ -0.031988 │ + └───────────┴───────────┴───────────┴───────────┴───────────┴───────────┘ + +With polars dataframes, the feature importance is stored in a dictionary instead of a +pandas Series, still sorted from the least to the most important feature: + +.. code:: python + + tr.feature_importances_ + +.. code:: text + + {'age': 41.41804062408145, + 's6': 64.76841724774016, + 's3': 113.96599187843395, + 's4': 182.174833735295, + 'sex': 238.6195264502995, + 'bp': 322.0918016880965, + 's2': 436.67158399913274, + 'bmi': 522.3301645404875, + 's5': 741.4713367752565, + 's1': 750.0238715216746} + And that's it! You now know how to select features by recursively removing them from a dataset. diff --git a/feature_engine/selection/recursive_feature_elimination.py b/feature_engine/selection/recursive_feature_elimination.py index 1bdc8a066..ad2c3ecb8 100644 --- a/feature_engine/selection/recursive_feature_elimination.py +++ b/feature_engine/selection/recursive_feature_elimination.py @@ -1,9 +1,10 @@ -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.model_selection import cross_validate from feature_engine._docstrings.fit_attributes import ( - _feature_importances_docstring, - _feature_importances_std_docstring, _feature_names_in_docstring, _n_features_in_docstring, _performance_drifts_docstring, @@ -28,6 +29,7 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from feature_engine.selection.base_selection_functions import _importance_series @Substitution( @@ -38,8 +40,6 @@ variables=_variables_numerical_docstring, confirm_variables=_confirm_variables_docstring, initial_model_performance_=_initial_model_performance_docstring, - feature_importances_=_feature_importances_docstring, - feature_importances_std_=_feature_importances_std_docstring, performance_drifts_=_performance_drifts_docstring, performance_drifts_std_=_performance_drifts_std_docstring, features_to_drop_=_features_to_drop_docstring, @@ -97,9 +97,14 @@ class RecursiveFeatureElimination(BaseRecursiveSelector): ---------- {initial_model_performance_} - {feature_importances_} + feature_importances_: + The feature importance (comes from step 2), sorted from the least to the + most important feature. A pandas Series with the features as index when X + is a pandas dataframe, and a dictionary with the features as keys otherwise. - {feature_importances_std_} + feature_importances_std_: + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. {performance_drifts_} @@ -144,31 +149,59 @@ class RecursiveFeatureElimination(BaseRecursiveSelector): 3 1 4 2 5 2 + + The same with polars: + + >>> import polars as pl + >>> rfe.fit_transform(pl.DataFrame(X.to_dict(orient="list")), y.to_list()) + shape: (6, 1) + ┌─────┐ + │ x2 │ + │ --- │ + │ i64 │ + ╞═════╡ + │ 2 │ + │ 4 │ + │ 3 │ + │ 1 │ + │ 2 │ + │ 2 │ + └─────┘ """ - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Find the important features. Note that the selector trains various models at each round of selection, so it might take a while. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. """ - X, y = super().fit(X, y) - - # Sort the feature importance values increasingly - self.feature_importances_.sort_values(ascending=True, inplace=True) + nw_X, y = super().fit(X, y) + + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + self.feature_importances_ = self.feature_importances_.sort_values() + else: + # numpy's default quicksort is the one of pandas' sort_values, so tied + # features are evaluated in the same order with every backend. + features = list(self.feature_importances_.keys()) + values = np.array(list(self.feature_importances_.values())) + order = values.argsort() + self.feature_importances_ = _importance_series( + X, [features[i] for i in order], values[order] + ) # to collect selected features _selected_features = [] - # temporary copy where we will remove features recursively - X_tmp = X[self.variables_].copy() + # features left in the model as we remove them recursively + remaining_features = list(self.variables_) # we need to update the performance as we remove features baseline_model_performance = self.initial_model_performance_ @@ -178,19 +211,25 @@ def fit(self, X: pd.DataFrame, y: pd.Series): self.performance_drifts_std_ = {} # evaluate every feature, starting from the least important - # remember that feature_importances_ is ordered already - for feature in list(self.feature_importances_.index): + for feature in self.feature_importances_.keys(): # if there is only 1 feature left - if X_tmp.shape[1] == 1: + if len(remaining_features) == 1: self.performance_drifts_[feature] = 0 _selected_features.append(feature) break # remove feature and train new model + features_tmp = [f for f in remaining_features if f != feature] + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + X_tmp = X[features_tmp] + else: + X_tmp = nw_X.select(nw.col(*features_tmp)).to_native() + model_tmp = cross_validate( estimator=self.estimator, - X=X_tmp.drop(columns=feature), + X=X_tmp, y=y, cv=self._cv, groups=self.groups, @@ -214,7 +253,7 @@ def fit(self, X: pd.DataFrame, y: pd.Series): else: # remove feature and adjust initial performance - X_tmp = X_tmp.drop(columns=feature) + remaining_features = features_tmp baseline_model = cross_validate( estimator=self.estimator, diff --git a/tests/test_selection/test_recursive_feature_elimination.py b/tests/test_selection/test_recursive_feature_elimination.py index 27eb689f9..3e56afb01 100644 --- a/tests/test_selection/test_recursive_feature_elimination.py +++ b/tests/test_selection/test_recursive_feature_elimination.py @@ -1,31 +1,79 @@ +import re + +import numpy as np import pandas as pd import pytest +from sklearn.datasets import load_diabetes from sklearn.ensemble import RandomForestClassifier from sklearn.linear_model import Lasso, LinearRegression, LogisticRegression -from sklearn.model_selection import KFold, GroupKFold +from sklearn.model_selection import GroupKFold, KFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier from sklearn.tree import DecisionTreeRegressor from feature_engine.selection import RecursiveFeatureElimination +from tests.backend_helpers import frame_to_dict, make_series + + +@pytest.fixture(scope="module") +def data_diabetes(): + X, y = load_diabetes(return_X_y=True) + data = {f"var_{i}": X[:, i].tolist() for i in range(10)} + data["target"] = y.tolist() + return data + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _rounded(drifts): + return {k: round(v, 4) for k, v in drifts.items()} + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RecursiveFeatureElimination(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (Lasso(), "neg_mean_squared_error", KFold(), [1, 2], 1, True), + (DecisionTreeRegressor(), "r2", StratifiedKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + sel = RecursiveFeatureElimination( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert sel.estimator is estimator + assert sel.scoring == scoring + assert sel.cv is cv + assert sel.groups == groups + assert sel.threshold == threshold + assert sel.confirm_variables is confirm_variables + -# tests for classification -_model_and_expectations = [ +# fit and transform +_classification_expectations = [ ( RandomForestClassifier(n_estimators=5, random_state=1), 3, 0.001, "roc_auc", - [ - "var_1", - "var_2", - "var_3", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], + ["var_0", "var_4"], { "var_5": -0.0, "var_3": 0.0009, @@ -34,7 +82,7 @@ "var_11": 0.001, "var_10": 0.0009, "var_8": 0.0001, - "var_0": 0.0019, + "var_0": 0.002, "var_9": 0.0, "var_6": -0.0, "var_7": -0.0015, @@ -46,17 +94,7 @@ 2, 0.0001, "accuracy", - [ - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_9", - "var_10", - "var_11", - ], + ["var_0", "var_7", "var_8"], { "var_2": 0.0, "var_9": 0.0, @@ -76,58 +114,44 @@ @pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, + "estimator, cv, threshold, scoring, selected, drifts", + _classification_expectations, ) def test_classification( - estimator, cv, threshold, scoring, dropped_features, performances, df_test + make_df, data_classification, estimator, cv, threshold, scoring, selected, drifts ): - X, y = df_test - + X, y = _split_target(make_df, data_classification) sel = RecursiveFeatureElimination( estimator=estimator, cv=cv, threshold=threshold, scoring=scoring ) + Xt = sel.fit_transform(X, y) - sel.fit(X, y) - - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) - - # test fit attrs - assert sel.features_to_drop_ == dropped_features - - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs.keys() == performances.keys() - for key in performances: - assert rounded_perfs[key] == pytest.approx(performances[key], abs=0.001) - - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert sel.features_to_drop_ == [f for f in X.columns if f not in selected] + # features are evaluated from the least to the most important + assert list(sel.performance_drifts_.keys()) == list(drifts.keys()) + assert _rounded(sel.performance_drifts_) == pytest.approx(drifts, abs=0.001) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_classification[f] for f in selected} -# tests for regression -_model_and_expectations = [ +_regression_expectations = [ ( Lasso(alpha=0.001, random_state=10), 3, 0.1, "r2", - [0, 1, 3, 4, 5, 6, 7, 9], + ["var_2", "var_8"], { - 0: -0.0032, - 9: -0.0003, - 6: -0.0008, - 7: 0.0001, - 1: 0.012, - 3: 0.0199, - 5: 0.0023, - 2: 0.1378, - 4: 0.0069, - 8: 0.115, + "var_0": -0.0032, + "var_9": -0.0003, + "var_6": -0.0008, + "var_7": 0.0001, + "var_1": 0.012, + "var_3": 0.0199, + "var_5": 0.0023, + "var_2": 0.1378, + "var_4": 0.0069, + "var_8": 0.115, }, ), ( @@ -135,135 +159,95 @@ def test_classification( 2, 100, "neg_mean_squared_error", - [1, 4], + ["var_0", "var_2", "var_3", "var_5", "var_6", "var_7", "var_8", "var_9"], { - 0: 481.9525, - 1: 64.6086, - 2: 1418.81, - 3: 345.2262, - 4: -200.8348, - 5: 438.2579, - 6: 286.6561, - 7: 246.7828, - 8: 301.6968, - 9: 700.138, + "var_1": 64.6086, + "var_4": -200.8348, + "var_0": 481.9525, + "var_6": 286.6561, + "var_9": 700.138, + "var_3": 345.2262, + "var_7": 246.7828, + "var_5": 438.2579, + "var_8": 301.6968, + "var_2": 1418.81, }, ), ] @pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, + "estimator, cv, threshold, scoring, selected, drifts", + _regression_expectations, ) def test_regression( - estimator, - cv, - threshold, - scoring, - dropped_features, - performances, - load_diabetes_dataset, + make_df, data_diabetes, estimator, cv, threshold, scoring, selected, drifts ): - # test for regression using cv=3, and the r2 as metric. - X, y = load_diabetes_dataset - + X, y = _split_target(make_df, data_diabetes) sel = RecursiveFeatureElimination( estimator=estimator, cv=cv, threshold=threshold, scoring=scoring ) + Xt = sel.fit_transform(X, y) - sel.fit(X, y) - - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) - - # test fit attrs - assert sel.features_to_drop_ == dropped_features + assert sel.features_to_drop_ == [f for f in X.columns if f not in selected] + assert list(sel.performance_drifts_.keys()) == list(drifts.keys()) + assert _rounded(sel.performance_drifts_) == pytest.approx(drifts) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_diabetes[f] for f in selected} - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == performances - - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) - - -def test_stops_when_only_one_feature_remains(): - linear_model = LinearRegression() - - # Feature x shows 100% correlation with target variable - # Feature x shows 0% correlation with target variable - # Target variable: y - - df = pd.DataFrame( - { - "x": [0, 0, 0, 0, 0, 1, 1, 1, 1, 1], - "z": [1, 1, 1, 1, 1, 1, 1, 1, 1, 1], - "y": [0, 0, 0, 0, 0, 1, 1, 1, 1, 1], - } - ) - - transformer = RecursiveFeatureElimination( - estimator=linear_model, scoring="r2", cv=3 - ) - output = transformer.fit_transform(df[["x", "z"]], df["y"]) - pd.testing.assert_frame_equal(output, df["x"].to_frame()) - -def test_performance_drift_std(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureElimination(estimator=linear_model, scoring="r2", cv=3) +def test_performance_drifts_and_std(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + sel = RecursiveFeatureElimination(LinearRegression(), scoring="r2", cv=3) sel.fit(X, y) - drifts = { - 0: -0.0033, - 9: -0.0003, - 6: -0.0007, - 7: 0.0001, - 1: 0.012, - 3: 0.0286, - 5: 0.0126, - 2: 0.0663, - 8: 0.1094, - 4: 0.0243, - } - - drfts_std = { - 0: 0.0136, - 9: 0.0168, - 6: 0.0169, - 7: 0.018, - 1: 0.0252, - 3: 0.0084, - 5: 0.0087, - 2: 0.0425, - 8: 0.0468, - 4: 0.0162, + assert sel.features_to_drop_ == ["var_0", "var_6", "var_7", "var_9"] + assert _rounded(sel.performance_drifts_) == { + "var_0": -0.0033, + "var_9": -0.0003, + "var_6": -0.0007, + "var_7": 0.0001, + "var_1": 0.012, + "var_3": 0.0286, + "var_5": 0.0126, + "var_2": 0.0663, + "var_8": 0.1094, + "var_4": 0.0243, } - - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ + assert _rounded(sel.performance_drifts_std_) == { + "var_0": 0.0136, + "var_9": 0.0168, + "var_6": 0.0169, + "var_7": 0.018, + "var_1": 0.0252, + "var_3": 0.0084, + "var_5": 0.0087, + "var_2": 0.0425, + "var_8": 0.0468, + "var_4": 0.0162, } - assert rounded_perfs == drifts - rounded_perfs = { - key: round(sel.performance_drifts_std_[key], 4) - for key in sel.performance_drifts_std_ - } - assert rounded_perfs == drfts_std - -def test_feature_importance(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureElimination(estimator=linear_model, scoring="r2", cv=3) +def test_feature_importances_are_sorted_ascending(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + sel = RecursiveFeatureElimination(LinearRegression(), scoring="r2", cv=3) sel.fit(X, y) - imps = [ + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(sel.feature_importances_, importance_type) + assert list(sel.feature_importances_.keys()) == [ + "var_0", + "var_9", + "var_6", + "var_7", + "var_1", + "var_3", + "var_5", + "var_2", + "var_8", + "var_4", + ] + assert [round(v, 2) for v in dict(sel.feature_importances_).values()] == [ 41.42, 64.77, 113.97, @@ -275,60 +259,155 @@ def test_feature_importance(load_diabetes_dataset): 741.47, 750.02, ] - imps_std = [18.22, 68.35, 86.03, 57.11, 329.38, 299.76, 72.81, 47.93, 117.83, 42.75] - - assert round(sel.feature_importances_, 2).to_list() == imps - assert round(sel.feature_importances_std_, 2).to_list() == imps_std + # the standard deviation keeps the order of the variables + assert list(sel.feature_importances_std_.keys()) == list(X.columns) + assert [round(v, 2) for v in dict(sel.feature_importances_std_).values()] == [ + 18.22, + 68.35, + 86.03, + 57.11, + 329.38, + 299.76, + 72.81, + 47.93, + 117.83, + 42.75, + ] -def test_cv_generator(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - cv = KFold(n_splits=3) - sel = RecursiveFeatureElimination(estimator=linear_model, scoring="r2", cv=3).fit( - X, y +def test_features_with_tied_importance_keep_their_order(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = RecursiveFeatureElimination( + Lasso(alpha=0.01, random_state=1), scoring="r2", threshold=-100 ) - expected = sel.transform(X) + sel.fit(X, y) - sel = RecursiveFeatureElimination(estimator=linear_model, scoring="r2", cv=cv).fit( - X, y + assert list(sel.feature_importances_.keys()) == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_9", + "var_10", + "var_11", + "var_8", + "var_7", + ] + assert list(sel.performance_drifts_.keys()) == list( + sel.feature_importances_.keys() ) - test1 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test1) - - sel = RecursiveFeatureElimination( - estimator=linear_model, scoring="r2", cv=cv.split(X, y) - ).fit(X, y) - test2 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test2) + assert sel.features_to_drop_ == [] + + +def test_stops_when_only_one_feature_remains(make_df): + # x is identical to the target and z is constant + x = [0, 0, 0, 0, 0, 1, 1, 1, 1, 1] + X = make_df({"x": x, "z": [1] * 10}) + sel = RecursiveFeatureElimination(LinearRegression(), scoring="r2", cv=3) + Xt = sel.fit_transform(X, make_series(make_df, x)) + + assert sel.features_to_drop_ == ["z"] + assert sel.performance_drifts_ == {"z": 0.0, "x": 0} + assert list(sel.performance_drifts_std_.keys()) == ["z"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"x": x} + + +def test_selection_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_ + X, y = _split_target(make_df, data_classification) + sel = RecursiveFeatureElimination(KNeighborsClassifier(), threshold=0.001) + Xt = sel.fit_transform(X, y) + + assert sel.features_to_drop_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_8", + "var_9", + "var_11", + ] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_7", "var_10"] -def test_recursive_feature_elimination_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_cv_splitter_and_generator(make_df, data_diabetes, cv_type): + X, y = _split_target(make_df, data_diabetes) + cv = KFold(n_splits=3) + if cv_type == "generator": + cv = cv.split(X, y) + sel = RecursiveFeatureElimination(LinearRegression(), scoring="r2", cv=cv) + sel.fit(X, y) - estimator = LinearRegression() - scoring = "neg_mean_absolute_error" + assert sel.features_to_drop_ == ["var_0", "var_6", "var_7", "var_9"] - sel_expected = RecursiveFeatureElimination( - estimator=estimator, - scoring=scoring, - cv=cv_indices, - ) - X_tr_expected = sel_expected.fit_transform(X, y) +def test_groups(make_df): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(5)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(10), 10) + X, y = _split_target(make_df, data) + cv = GroupKFold(n_splits=3) + sel_splits = RecursiveFeatureElimination( + LinearRegression(), + scoring="neg_mean_absolute_error", + cv=cv.split(X, y, groups=groups), + ) + sel_splits.fit(X, y) sel = RecursiveFeatureElimination( - estimator=estimator, - scoring=scoring, - cv=cv, - groups=groups, + LinearRegression(), scoring="neg_mean_absolute_error", cv=cv, groups=groups ) + sel.fit(X, y) + + assert sel.features_to_drop_ == sel_splits.features_to_drop_ + assert sel.performance_drifts_ == pytest.approx(sel_splits.performance_drifts_) + - X_tr = sel.fit_transform(X, y) +def test_only_numerical_variables_are_evaluated(make_df, data_classification): + data = {f"var_{i}": data_classification[f"var_{i}"] for i in [0, 4, 7]} + data["cat"] = ["a", "b"] * 500 + data["target"] = data_classification["target"] + X, y = _split_target(make_df, data) + sel = RecursiveFeatureElimination(LogisticRegression(), threshold=0.0005) + Xt = sel.fit_transform(X, y) - pd.testing.assert_frame_equal( - X_tr_expected, - X_tr, + assert sel.variables_ == ["var_0", "var_4", "var_7"] + assert sel.features_to_drop_ == ["var_4"] + assert frame_to_dict(Xt) == { + "var_0": data["var_0"], + "cat": data["cat"], + "var_7": data["var_7"], + } + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_list_and_array_target(make_df, data_diabetes, target_type): + X, _ = _split_target(make_df, data_diabetes) + y = target_type(data_diabetes["target"]) + sel = RecursiveFeatureElimination(LinearRegression(), scoring="r2", cv=3) + sel.fit(X, y) + + assert sel.features_to_drop_ == ["var_0", "var_6", "var_7", "var_9"] + + +def test_integer_column_names(load_diabetes_dataset): + X, y = load_diabetes_dataset + sel = RecursiveFeatureElimination( + Lasso(alpha=0.001, random_state=10), cv=3, threshold=0.1, scoring="r2" ) + Xt = sel.fit_transform(X, y) + + assert sel.features_to_drop_ == [0, 1, 3, 4, 5, 6, 7, 9] + assert list(sel.feature_importances_.index) == [0, 9, 6, 7, 1, 3, 5, 2, 4, 8] + assert list(sel.performance_drifts_.keys()) == [0, 9, 6, 7, 1, 3, 5, 2, 4, 8] + pd.testing.assert_frame_equal(Xt, X[[2, 8]]) diff --git a/tests/test_selection/test_recursive_feature_selectors.py b/tests/test_selection/test_recursive_feature_selectors.py index d9d3aebc8..651b55355 100644 --- a/tests/test_selection/test_recursive_feature_selectors.py +++ b/tests/test_selection/test_recursive_feature_selectors.py @@ -10,13 +10,9 @@ from sklearn.svm import SVR from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor -from feature_engine.selection import ( - RecursiveFeatureAddition, - RecursiveFeatureElimination, -) +from feature_engine.selection import RecursiveFeatureAddition _selectors = [ - RecursiveFeatureElimination, RecursiveFeatureAddition, ] @@ -164,12 +160,6 @@ def test_feature_importances(_estimator, _importance, df_test): list(np.round(sel.feature_importances_.values, 2)) == np.round(_importance, 2) ).all() - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) - _importance.sort(reverse=False) - assert ( - list(np.round(sel.feature_importances_.values, 2)) == np.round(_importance, 2) - ).all() - ests = [ BaggingClassifier( @@ -206,11 +196,6 @@ def test_permutation_importance(_estimator, df_test): sel.feature_importances_.sort_index(), expected_importances.sort_index() ) - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) - assert_series_equal( - sel.feature_importances_.sort_index(), expected_importances.sort_index() - ) - @pytest.mark.parametrize("_estimator", ests) def test_selection_after_permutation_importance(_estimator, df_test): @@ -218,7 +203,3 @@ def test_selection_after_permutation_importance(_estimator, df_test): sel = RecursiveFeatureAddition(_estimator) Xtr = sel.fit_transform(X, y) assert Xtr.shape[1] < X.shape[1] - - sel = RecursiveFeatureElimination(_estimator) - Xtr = sel.fit_transform(X, y) - assert Xtr.shape[1] < X.shape[1]