diff --git a/docs/user_guide/selection/RecursiveFeatureAddition.rst b/docs/user_guide/selection/RecursiveFeatureAddition.rst index ee1e2f4a0..90287033c 100644 --- a/docs/user_guide/selection/RecursiveFeatureAddition.rst +++ b/docs/user_guide/selection/RecursiveFeatureAddition.rst @@ -136,7 +136,7 @@ entire dataset: .. code:: python - 0.488702767247119 + np.float64(0.48870212980353145) Evaluating feature importance ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -202,15 +202,15 @@ In the following output we see the changes in performance returned by adding eac .. code:: python {'s1': 0, - 's5': 0.28371458794131676, - 'bmi': 0.1377714799388745, - 's2': 0.0023327265047610735, - 'bp': 0.018759914615172735, - 'sex': 0.0027996354657459643, - 's4': 0.002695149440021638, - 's3': 0.002683934134630306, - 's6': 0.000304067408860742, - 'age': -0.007387230783454768} + 's5': np.float64(0.28371458794131676), + 'bmi': np.float64(0.13777147993887456), + 's2': np.float64(0.0023327265047610735), + 'bp': np.float64(0.01875991461517268), + 'sex': np.float64(0.002799635465745798), + 's4': np.float64(0.002695149440021638), + 's3': np.float64(0.0026839341346303613), + 's6': np.float64(0.0003040674088605755), + 'age': np.float64(-0.007387230783454768)} We can also check out the standard deviation of the performance drift: @@ -225,15 +225,15 @@ returned by adding each feature: .. code:: python {'s1': 0, - 's5': 0.029336910701570382, - 'bmi': 0.01752426732750277, - 's2': 0.020525965661877265, - 'bp': 0.017326401244547558, - 'sex': 0.00867675077259389, - 's4': 0.024234566449074676, - 's3': 0.023391851139598106, - 's6': 0.016865740401721313, - 'age': 0.02042081611218045} + 's5': np.float64(0.02933691070157033), + 'bmi': np.float64(0.017524267327502716), + 's2': np.float64(0.020525965661877265), + 'bp': np.float64(0.017326401244547592), + 'sex': np.float64(0.008676750772593802), + 's4': np.float64(0.024234566449074697), + 's3': np.float64(0.02339185113959813), + 's6': np.float64(0.01686574040172137), + 'age': np.float64(0.020420816112180475)} We can now plot the performance change with the standard deviation to identify importance features: @@ -321,6 +321,67 @@ be dropped: [False, False, True, True, True, False, False, False, True, False] +With polars +~~~~~~~~~~~ + +:class:`RecursiveFeatureAddition` also selects features from polars dataframes, and +returns a polars dataframe. Let's load the diabetes dataset into a polars dataframe: + +.. code:: python + + import polars as pl + + diabetes = load_diabetes() + X = pl.DataFrame(diabetes.data, schema=diabetes.feature_names) + y = pl.Series("target", diabetes.target) + +Now, we select features as we did with pandas: + +.. code:: python + + tr = RecursiveFeatureAddition(estimator=LinearRegression(), scoring="r2", cv=3) + Xt = tr.fit_transform(X, y) + print(Xt.head()) + +The same 4 features are retained: + +.. code:: text + + shape: (5, 4) + ┌───────────┬───────────┬───────────┬───────────┐ + │ bmi ┆ bp ┆ s1 ┆ s5 │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╪═══════════╡ + │ 0.061696 ┆ 0.021872 ┆ -0.044223 ┆ 0.019907 │ + │ -0.051474 ┆ -0.026328 ┆ -0.008449 ┆ -0.068332 │ + │ 0.044451 ┆ -0.00567 ┆ -0.045599 ┆ 0.002861 │ + │ -0.011595 ┆ -0.036656 ┆ 0.012191 ┆ 0.022688 │ + │ -0.036385 ┆ 0.021872 ┆ 0.003935 ┆ -0.031988 │ + └───────────┴───────────┴───────────┴───────────┘ + +With polars, the feature importance and its standard deviation are dictionaries instead +of pandas Series: + +.. code:: python + + tr.feature_importances_ + +In the following output we see the features sorted by their importance: + +.. code:: python + + {'s1': 750.0238715216746, + 's5': 741.4713367752565, + 'bmi': 522.3301645404875, + 's2': 436.67158399913274, + 'bp': 322.0918016880965, + 'sex': 238.6195264502995, + 's4': 182.174833735295, + 's3': 113.96599187843395, + 's6': 64.76841724774016, + 'age': 41.41804062408145} + And that's it! You now know how to select features by recursively adding them to a dataset. Additional resources diff --git a/feature_engine/_docstrings/fit_attributes.py b/feature_engine/_docstrings/fit_attributes.py index 3a91a61bd..22a2b3746 100644 --- a/feature_engine/_docstrings/fit_attributes.py +++ b/feature_engine/_docstrings/fit_attributes.py @@ -35,11 +35,14 @@ # used by selection module _feature_importances_docstring = """feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. """.rstrip() _feature_importances_std_docstring = """feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. """.rstrip() _performance_drifts_docstring = """performance_drifts_: diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..40e84f493 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,47 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] - - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) - - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" - - self.feature_names_in_ = X.columns.to_list() + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + if len(features) == 0: + # nw.col() needs at least one name. polars frames without columns + # have no rows either. + return nw_X.select([]).to_native() + return nw_X.select(nw.col(*features)).to_native() + + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" + + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/selection/recursive_feature_addition.py b/feature_engine/selection/recursive_feature_addition.py index c54185e04..a5247d23e 100644 --- a/feature_engine/selection/recursive_feature_addition.py +++ b/feature_engine/selection/recursive_feature_addition.py @@ -1,4 +1,9 @@ -import pandas as pd +from typing import Any, Dict + +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.model_selection import cross_validate from feature_engine._docstrings.fit_attributes import ( @@ -28,6 +33,7 @@ ) from feature_engine._docstrings.substitute import Substitution from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from feature_engine.selection.base_selection_functions import _importance_series @Substitution( @@ -144,85 +150,84 @@ class RecursiveFeatureAddition(BaseRecursiveSelector): 3 1 1 4 2 0 5 2 1 + + With polars: + + >>> import polars as pl + >>> from sklearn.ensemble import RandomForestClassifier + >>> from feature_engine.selection import RecursiveFeatureAddition + >>> X = pl.DataFrame(dict(x1 = [1000,2000,1000,1000,2000,3000], + >>> x2 = [2,4,3,1,2,2], + >>> x3 = [1,1,1,0,0,0], + >>> x4 = [1,2,1,1,0,1], + >>> x5 = [1,1,1,1,1,1])) + >>> y = pl.Series([1,0,0,1,1,0]) + >>> rfa = RecursiveFeatureAddition(RandomForestClassifier(random_state=42), cv=2) + >>> rfa.fit_transform(X, y) + shape: (6, 2) + ┌─────┬─────┐ + │ x2 ┆ x4 │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞═════╪═════╡ + │ 2 ┆ 1 │ + │ 4 ┆ 2 │ + │ 3 ┆ 1 │ + │ 1 ┆ 1 │ + │ 2 ┆ 0 │ + │ 2 ┆ 1 │ + └─────┴─────┘ """ - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries): """ Find the important features. Note that the selector trains various models at each round of selection, so it might take a while. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The input dataframe + X: dataframe of shape = [n_samples, n_features] + The input dataframe. y: array-like of shape (n_samples) Target variable. Required to train the estimator. """ - X, y = super().fit(X, y) - - # Sort the feature importance values decreasingly - self.feature_importances_.sort_values(ascending=False, inplace=True) - - # Extract most important feature from the ordered list of features - first_most_important_feature = list(self.feature_importances_.index)[0] - - # Run baseline model using only the most important feature - baseline_model = cross_validate( - estimator=self.estimator, - X=X[first_most_important_feature].to_frame(), - y=y, - cv=self._cv, - groups=self.groups, - scoring=self.scoring, - return_estimator=True, + nw_X, y = super().fit(X, y) + + importances = dict(self.feature_importances_) + features = list(importances) + values = np.array(list(importances.values())) + # Same order as pandas' sort_values(ascending=False), which sorts the reversed + # values, so features with the same importance are ranked as before. + order = (len(values) - 1 - values[::-1].argsort(kind="quicksort"))[::-1] + ranked_features = [features[i] for i in order] + self.feature_importances_ = _importance_series( + X, ranked_features, values[order] ) - # Save baseline model performance - baseline_model_performance = baseline_model["test_score"].mean() + first_most_important_feature = ranked_features[0] + baseline_scores = self._cross_validate( + X, nw_X, y, [first_most_important_feature] + ) + baseline_model_performance = baseline_scores.mean() - # list to collect selected features - # It is initialized with the most important feature _selected_features = [first_most_important_feature] - - # dict to collect features and their performance_drift - # It is initialized with the performance drift of - # the most important feature - self.performance_drifts_ = {first_most_important_feature: 0} - self.performance_drifts_std_ = {first_most_important_feature: 0} - - # loop over the ordered list of features by feature importance starting - # from the second element in the list. - for feature in list(self.feature_importances_.index)[1:]: - - # Add feature and train new model - model_tmp = cross_validate( - estimator=self.estimator, - X=X[_selected_features + [feature]], - y=y, - cv=self._cv, - groups=self.groups, - scoring=self.scoring, - return_estimator=True, - ) - - # assign new model performance - model_tmp_performance = model_tmp["test_score"].mean() - - # Calculate performance drift + self.performance_drifts_: Dict[Any, float] = {first_most_important_feature: 0} + self.performance_drifts_std_: Dict[Any, float] = { + first_most_important_feature: 0 + } + + for feature in ranked_features[1:]: + scores = self._cross_validate(X, nw_X, y, _selected_features + [feature]) + model_tmp_performance = scores.mean() performance_drift = model_tmp_performance - baseline_model_performance - # Save feature and performance drift self.performance_drifts_[feature] = performance_drift - self.performance_drifts_std_[feature] = model_tmp["test_score"].std() + self.performance_drifts_std_[feature] = scores.std() - # If new performance model is if performance_drift > self.threshold: - # add feature to the list of selected features _selected_features.append(feature) - - # Update new baseline model performance baseline_model_performance = model_tmp_performance self.features_to_drop_ = [ @@ -230,3 +235,22 @@ def fit(self, X: pd.DataFrame, y: pd.Series): ] return self + + def _cross_validate( + self, X: IntoDataFrame, nw_X: nw.DataFrame, y: IntoSeries, features: list + ) -> np.ndarray: + """Return the test scores of the estimator trained on the features.""" + if nwd.is_pandas_dataframe(X) is True: + # pandas is faster than narwhals. + X_model = X[features] + else: + X_model = nw_X.select(nw.col(*features)).to_native() + + return cross_validate( + estimator=self.estimator, + X=X_model, + y=y, + cv=self._cv, + groups=self.groups, + scoring=self.scoring, + )["test_score"] diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..64e43b88c 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,164 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + (["Name", "City", "Age", "Marks", "dob"], []), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) diff --git a/tests/test_selection/test_recursive_feature_addition.py b/tests/test_selection/test_recursive_feature_addition.py index 0acd558c1..fdef405fa 100644 --- a/tests/test_selection/test_recursive_feature_addition.py +++ b/tests/test_selection/test_recursive_feature_addition.py @@ -1,310 +1,602 @@ +import re + +import numpy as np import pandas as pd import pytest -from sklearn.ensemble import RandomForestClassifier -from sklearn.linear_model import Lasso, LogisticRegression, LinearRegression -from sklearn.model_selection import KFold, GroupKFold +from sklearn.datasets import load_diabetes, make_classification +from sklearn.ensemble import HistGradientBoostingClassifier, RandomForestClassifier +from sklearn.exceptions import NotFittedError +from sklearn.linear_model import Lasso, LinearRegression, LogisticRegression +from sklearn.model_selection import GroupKFold, KFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier from sklearn.tree import DecisionTreeRegressor from feature_engine.selection import RecursiveFeatureAddition +from tests.backend_helpers import frame_to_dict, make_series -# tests for classification -_model_and_expectations = [ - ( - RandomForestClassifier(n_estimators=5, random_state=1), - 3, - 0.001, - "roc_auc", - [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", - ], - { - "var_4": 0, - "var_7": 0.0241, - "var_6": -0.001, - "var_9": -0.001, - "var_0": -0.0, - "var_8": -0.0011, - "var_10": -0.0011, - "var_11": -0.001, - "var_1": -0.0, - "var_2": -0.0001, - "var_3": -0.0011, - "var_5": -0.0001, - }, - ), - ( - LogisticRegression(random_state=10), - 2, - 0.0001, - "accuracy", - [ - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_9", - "var_10", - "var_11", - ], - { - "var_7": 0, - "var_8": 0.001, - "var_0": 0.002, - "var_6": 0.0, - "var_4": 0.0, - "var_11": -0.001, - "var_1": -0.001, - "var_5": -0.003, - "var_3": -0.002, - "var_10": 0.0, - "var_9": 0.0, - "var_2": 0.0, - }, - ), -] +@pytest.fixture(scope="module") +def data_diabetes(): + X, y = load_diabetes(return_X_y=True, as_frame=True) + data = {col: X[col].tolist() for col in X.columns} + data["target"] = y.tolist() + return data -@pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, -) -def test_classification( - estimator, cv, threshold, scoring, dropped_features, performances, df_test -): - X, y = df_test - sel = RecursiveFeatureAddition( - estimator=estimator, cv=cv, threshold=threshold, scoring=scoring - ) +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + X = rng.normal(size=(100, 5)) + data = {f"var_{i}": X[:, i].tolist() for i in range(5)} + data["target"] = (3 * X[:, 0] + X[:, 1] + rng.normal(size=100)).tolist() + groups = np.repeat(np.arange(10), 10) + rng.shuffle(groups) + return data, groups - sel.fit(X, y) - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) - # test fit attrs - assert sel.features_to_drop_ == dropped_features - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == performances +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + RecursiveFeatureAddition(RandomForestClassifier(), threshold=threshold) - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +@pytest.mark.parametrize("confirm_variables", [None, 1, "True", [True]]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + RecursiveFeatureAddition( + RandomForestClassifier(), confirm_variables=confirm_variables + ) -# tests for regression -_model_and_expectations = [ - ( - Lasso(alpha=0.001, random_state=10), - 3, - 0.1, - "r2", - [0, 1, 3, 4, 5, 6, 7, 9], - { - 8: 0, - 4: 0.0059, - 2: 0.1367, - 5: -0.0026, - 3: 0.0177, - 1: -0.0045, - 7: -0.0035, - 6: 0.0088, - 9: 0.002, - 0: -0.0114, - }, - ), - ( - DecisionTreeRegressor(random_state=10), - 2, - 100, - "neg_mean_squared_error", - [0, 3, 4, 5, 6, 7, 8, 9], - { - 0: -1693.8544, - 1: 106.9272, - 2: 0, - 3: -222.2945, - 4: -781.6593, - 5: -943.6299, - 6: -1701.2939, - 7: 99.315, - 8: -660.1107, - 9: -716.6378, - }, - ), -] +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (Lasso(), "neg_mean_squared_error", KFold(), [1, 2], 1, True), + (DecisionTreeRegressor(), "r2", StratifiedKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = RecursiveFeatureAddition( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + +# fit and transform @pytest.mark.parametrize( - "estimator, cv, threshold, scoring, dropped_features, performances", - _model_and_expectations, + "estimator, cv, threshold, scoring, performance, drifts, dropped, retained", + [ + ( + RandomForestClassifier(n_estimators=5, random_state=1), + 3, + 0.001, + "roc_auc", + 0.9954973573949477, + { + "var_4": 0, + "var_7": 0.024094900525623353, + "var_6": -0.0010340785943196984, + "var_9": -0.0010221260221260353, + "var_0": -1.2604530676751935e-05, + "var_8": -0.001076745655058886, + "var_10": -0.0011244472840857833, + "var_11": -0.001028283407801478, + "var_1": -1.832727736339468e-05, + "var_2": -9.015137027179598e-05, + "var_3": -0.001076636995311686, + "var_5": -7.218629206573457e-05, + }, + [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_5", + "var_6", + "var_8", + "var_9", + "var_10", + "var_11", + ], + ["var_4", "var_7"], + ), + ( + LogisticRegression(random_state=10), + 2, + 0.0001, + "accuracy", + 0.99, + { + "var_7": 0, + "var_8": 0.001, + "var_0": 0.002, + "var_6": 0, + "var_4": 0, + "var_11": -0.001, + "var_1": -0.001, + "var_5": -0.003, + "var_3": -0.002, + "var_10": 0, + "var_9": 0, + "var_2": 0, + }, + [ + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_9", + "var_10", + "var_11", + ], + ["var_0", "var_7", "var_8"], + ), + ], ) -def test_regression( +def test_classification( + make_df, + data_classification, estimator, cv, threshold, scoring, - dropped_features, - performances, - load_diabetes_dataset, + performance, + drifts, + dropped, + retained, ): - # test for regression using cv=3, and the r2 as metric. - X, y = load_diabetes_dataset - - sel = RecursiveFeatureAddition( + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( estimator=estimator, cv=cv, threshold=threshold, scoring=scoring ) + Xt = selector.fit_transform(X, y) - sel.fit(X, y) + assert selector.initial_model_performance_ == pytest.approx(performance) + assert selector.performance_drifts_ == pytest.approx(drifts) + assert list(selector.performance_drifts_) == list(drifts) + assert selector.features_to_drop_ == dropped + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_classification[f] for f in retained} - Xtransformed = X.copy() - Xtransformed = Xtransformed.drop(labels=dropped_features, axis=1) - # test fit attrs - assert sel.features_to_drop_ == dropped_features +@pytest.mark.parametrize( + "estimator, cv, threshold, scoring, drifts, dropped, retained", + [ + ( + Lasso(alpha=0.001, random_state=10), + 3, + 0.1, + "r2", + { + "s5": 0, + "s1": 0.005876977602335132, + "bmi": 0.1367220291124882, + "s2": -0.002593999790115431, + "bp": 0.017686140335452738, + "sex": -0.004478777508460652, + "s4": -0.003491685019777313, + "s3": 0.008817074774180478, + "s6": 0.002042365258727974, + "age": -0.011362436906615925, + }, + ["age", "sex", "bp", "s1", "s2", "s3", "s4", "s6"], + ["bmi", "s5"], + ), + ( + DecisionTreeRegressor(random_state=10), + 2, + 100, + "neg_mean_squared_error", + { + "bmi": 0, + "s5": -660.1106667923586, + "s2": -943.6298975615891, + "s4": 99.31498680241202, + "bp": -222.29449032177035, + "s6": -716.6378161136254, + "s3": -1701.2939247109098, + "age": -1693.8544450729005, + "s1": -781.6593093262954, + "sex": 106.92720965309127, + }, + ["age", "bp", "s1", "s2", "s3", "s4", "s5", "s6"], + ["sex", "bmi"], + ), + ], +) +def test_regression( + make_df, data_diabetes, estimator, cv, threshold, scoring, drifts, dropped, retained +): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition( + estimator=estimator, cv=cv, threshold=threshold, scoring=scoring + ) + Xt = selector.fit_transform(X, y) - assert len(sel.performance_drifts_.keys()) == len(X.columns) - assert all([var in sel.performance_drifts_.keys() for var in X.columns]) - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == performances - - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) - - -def test_performance_drift_std(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3) - sel.fit(X, y) - - drifts = { - 4: 0, - 8: 0.2837, - 2: 0.1378, - 5: 0.0023, - 3: 0.0188, - 1: 0.0028, - 7: 0.0027, - 6: 0.0027, - 9: 0.0003, - 0: -0.0074, - } + assert selector.performance_drifts_ == pytest.approx(drifts) + assert list(selector.performance_drifts_) == list(drifts) + assert selector.features_to_drop_ == dropped + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_diabetes[f] for f in retained} - drfts_std = { - 4: 0, - 8: 0.0293, - 2: 0.0175, - 5: 0.0205, - 3: 0.0173, - 1: 0.0087, - 7: 0.0242, - 6: 0.0234, - 9: 0.0169, - 0: 0.0204, - } - rounded_perfs = { - key: round(sel.performance_drifts_[key], 4) for key in sel.performance_drifts_ - } - assert rounded_perfs == drifts +def test_attributes_with_linear_regression(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) - rounded_perfs = { - key: round(sel.performance_drifts_std_[key], 4) - for key in sel.performance_drifts_std_ - } - assert rounded_perfs == drfts_std - - -def test_feature_importance(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3) - sel.fit(X, y) - - imps = [ - 750.02, - 741.47, - 522.33, - 436.67, - 322.09, - 238.62, - 182.17, - 113.97, - 64.77, - 41.42, + assert selector.initial_model_performance_ == pytest.approx(0.48870212980353145) + # the importance is sorted from the most to the least important feature. + assert dict(selector.feature_importances_) == pytest.approx( + { + "s1": 750.0238715216746, + "s5": 741.4713367752565, + "bmi": 522.3301645404875, + "s2": 436.67158399913274, + "bp": 322.0918016880965, + "sex": 238.6195264502995, + "s4": 182.174833735295, + "s3": 113.96599187843395, + "s6": 64.76841724774016, + "age": 41.41804062408145, + } + ) + assert list(selector.feature_importances_.keys()) == [ + "s1", + "s5", + "bmi", + "s2", + "bp", + "sex", + "s4", + "s3", + "s6", + "age", ] - imps_std = [18.22, 68.35, 86.03, 57.11, 329.38, 299.76, 72.81, 47.93, 117.83, 42.75] + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "age": 18.21715207624677, + "sex": 68.35471931174253, + "bmi": 86.03069812358049, + "bp": 57.11038282196886, + "s1": 329.3758185655728, + "s2": 299.7569984100255, + "s3": 72.80549636690168, + "s4": 47.925822175574034, + "s5": 117.8299487852095, + "s6": 42.75477362271488, + } + ) + assert selector.performance_drifts_ == pytest.approx( + { + "s1": 0, + "s5": 0.28371458794131676, + "bmi": 0.13777147993887456, + "s2": 0.0023327265047611845, + "bp": 0.018759914615172624, + "sex": 0.0027996354657458533, + "s4": 0.0026951494400216935, + "s3": 0.002683934134630417, + "s6": 0.00030406740886079753, + "age": -0.007387230783454712, + } + ) + assert selector.performance_drifts_std_ == pytest.approx( + { + "s1": 0, + "s5": 0.02933691070157033, + "bmi": 0.017524267327502716, + "s2": 0.020525965661877258, + "bp": 0.01732640124454753, + "sex": 0.008676750772593741, + "s4": 0.024234566449074697, + "s3": 0.02339185113959813, + "s6": 0.016865740401721358, + "age": 0.020420816112180495, + } + ) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] - assert round(sel.feature_importances_, 2).to_list() == imps - assert round(sel.feature_importances_std_, 2).to_list() == imps_std +def test_feature_importances_type(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) -def test_cv_generator(load_diabetes_dataset): - X, y = load_diabetes_dataset - linear_model = LinearRegression() - cv = KFold(n_splits=3) - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=3).fit(X, y) - expected = sel.transform(X) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) - sel = RecursiveFeatureAddition(estimator=linear_model, scoring="r2", cv=cv).fit( - X, y + +def test_ranking_of_features_with_equal_importance(make_df): + # Lasso sets most coefficients to 0. With more than 16 features the sort is not + # stable, so both backends must rank these ties as pandas' sort_values does. + X, y = make_classification( + n_samples=500, n_features=25, n_informative=4, random_state=3 + ) + X = make_df({f"var_{i}": X[:, i] for i in range(25)}) + selector = RecursiveFeatureAddition(Lasso(alpha=0.05), scoring="r2", cv=3) + selector.fit(X, make_series(make_df, y)) + + assert list(selector.feature_importances_.keys()) == [ + "var_22", + "var_13", + "var_16", + "var_9", + "var_19", + "var_11", + "var_0", + "var_14", + "var_23", + "var_21", + "var_20", + "var_18", + "var_17", + "var_15", + "var_12", + "var_1", + "var_10", + "var_8", + "var_7", + "var_6", + "var_5", + "var_4", + "var_3", + "var_2", + "var_24", + ] + assert list(selector.performance_drifts_) == list( + selector.feature_importances_.keys() ) - test1 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test1) - sel = RecursiveFeatureAddition( - estimator=linear_model, scoring="r2", cv=cv.split(X, y) - ).fit(X, y) - test2 = sel.transform(X) - pd.testing.assert_frame_equal(expected, test2) +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so the importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( + KNeighborsClassifier(), scoring="accuracy", cv=3 + ) + Xt = selector.fit_transform(X, y) -def test_recursive_feature_addition_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) - threshold = 0.5 + assert list(selector.feature_importances_.keys())[:3] == [ + "var_7", + "var_6", + "var_4", + ] + assert selector.feature_importances_["var_7"] == pytest.approx(0.2313333333333333) + assert selector.features_to_drop_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_6", + "var_8", + "var_9", + "var_10", + "var_11", + ] + assert frame_to_dict(Xt) == {"var_7": data_classification["var_7"]} - estimator = LinearRegression() - scoring = "neg_mean_absolute_error" - sel_expected = RecursiveFeatureAddition( - estimator=estimator, - scoring=scoring, - cv=cv_indices, - threshold=threshold, +def test_variables_and_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = RecursiveFeatureAddition( + LogisticRegression(), + cv=3, + variables=["var_0", "var_4", "var_7", "var_9", "Hola"], + confirm_variables=True, ) + Xt = selector.fit_transform(X, y) - X_tr_expected = sel_expected.fit_transform(X, y) + assert selector.variables_ == ["var_0", "var_4", "var_7", "var_9"] + assert selector.performance_drifts_ == pytest.approx( + { + "var_7": 0, + "var_0": 0.0007903910012343474, + "var_4": 0.0004547772620060453, + "var_9": 0.0005748100627617214, + } + ) + assert selector.features_to_drop_ == ["var_0", "var_4", "var_9"] + assert list(Xt.columns) == [ + "var_1", + "var_2", + "var_3", + "var_5", + "var_6", + "var_7", + "var_8", + "var_10", + "var_11", + ] - sel = RecursiveFeatureAddition( - estimator=estimator, - scoring=scoring, - cv=cv, + +def test_non_numerical_variables_are_ignored_and_kept(make_df, data_classification): + data = { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "var_7": data_classification["var_7"], + "target": data_classification["target"], + } + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition(LogisticRegression(), cv=3) + Xt = selector.fit_transform(X, y) + + assert selector.variables_ == ["var_0", "var_4", "var_7"] + assert list(selector.performance_drifts_) == ["var_7", "var_0", "var_4"] + assert selector.features_to_drop_ == ["var_0", "var_4"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"cat": data["cat"], "var_7": data["var_7"]} + + +def test_fit_with_missing_values(make_df, data_classification): + # The selector doesn't check for NaN: estimators that handle it can be used. + data = dict(data_classification) + for var in ["var_0", "var_4", "var_7"]: + data[var] = [None if i % 20 == 0 else v for i, v in enumerate(data[var])] + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition( + HistGradientBoostingClassifier(max_iter=10, random_state=0), + scoring="accuracy", + cv=3, + ) + Xt = selector.fit_transform(X, y) + + assert selector.features_to_drop_ == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_4", + "var_5", + "var_8", + "var_9", + "var_10", + "var_11", + ] + assert frame_to_dict(Xt) == {f: data[f] for f in ["var_6", "var_7"]} + + +def test_transform_returns_features_in_training_order(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + Xt = selector.transform(X[list(X.columns)[::-1]]) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {f: data_diabetes[f] for f in ["bmi", "bp", "s1", "s5"]} + + +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_cv_splitter_and_generator(make_df, data_diabetes, cv_type): + X, y = _split_target(make_df, data_diabetes) + cv = KFold(n_splits=3) + if cv_type == "generator": + cv = cv.split(X, y) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=cv) + selector.fit(X, y) + + assert selector.performance_drifts_["s5"] == pytest.approx(0.28371458794131676) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] + + +def test_cv_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + selector = RecursiveFeatureAddition( + LinearRegression(), + scoring="r2", + cv=GroupKFold(n_splits=3), groups=groups, - threshold=threshold, + threshold=0.05, + ) + selector.fit(X, y) + + assert selector.performance_drifts_ == pytest.approx( + { + "var_0": 0, + "var_1": 0.07658061969570773, + "var_4": 0.0015218369914395957, + "var_2": -0.009235024059350394, + "var_3": -0.002472987677324623, + } + ) + assert selector.features_to_drop_ == ["var_2", "var_3", "var_4"] + + indices = GroupKFold(n_splits=3).split(X, y, groups) + selector_indices = RecursiveFeatureAddition( + LinearRegression(), scoring="r2", cv=indices, threshold=0.05 ) - X_tr = sel.fit_transform(X, y) + selector_indices.fit(X, y) - pd.testing.assert_frame_equal( - X_tr_expected, - X_tr, + assert selector_indices.performance_drifts_ == pytest.approx( + selector.performance_drifts_ ) + assert selector_indices.features_to_drop_ == ["var_2", "var_3", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_list_and_array_target(make_df, data_diabetes, target_type): + X, _ = _split_target(make_df, data_diabetes) + y = target_type(data_diabetes["target"]) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + + assert selector.performance_drifts_["s5"] == pytest.approx(0.28371458794131676) + assert selector.features_to_drop_ == ["age", "sex", "s2", "s3", "s4", "s6"] + + +def test_error_if_transform_before_fit(make_df, data_diabetes): + X, _ = _split_target(make_df, data_diabetes) + msg = ( + "This RecursiveFeatureAddition instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + RecursiveFeatureAddition(LinearRegression()).transform(X) + + +def test_feature_importances_are_pandas_series(data_diabetes): + X, y = _split_target(pd.DataFrame, data_diabetes) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [ + 750.0238715216746, + 741.4713367752565, + 522.3301645404875, + 436.67158399913274, + 322.0918016880965, + 238.6195264502995, + 182.174833735295, + 113.96599187843395, + 64.76841724774016, + 41.41804062408145, + ], + index=["s1", "s5", "bmi", "s2", "bp", "sex", "s4", "s3", "s6", "age"], + ), + ) + + +def test_integer_column_names(data_diabetes): + X, y = _split_target(pd.DataFrame, data_diabetes) + X.columns = list(range(10)) + selector = RecursiveFeatureAddition(LinearRegression(), scoring="r2", cv=3) + Xt = selector.fit_transform(X, y) + + assert selector.features_to_drop_ == [0, 1, 5, 6, 7, 9] + assert list(selector.performance_drifts_) == [4, 8, 2, 5, 3, 1, 7, 6, 9, 0] + pd.testing.assert_frame_equal(Xt, X[[2, 3, 4, 8]]) diff --git a/tests/test_selection/test_recursive_feature_selectors.py b/tests/test_selection/test_recursive_feature_selectors.py index d9d3aebc8..239bb2591 100644 --- a/tests/test_selection/test_recursive_feature_selectors.py +++ b/tests/test_selection/test_recursive_feature_selectors.py @@ -10,14 +10,10 @@ from sklearn.svm import SVR from sklearn.tree import DecisionTreeClassifier, DecisionTreeRegressor -from feature_engine.selection import ( - RecursiveFeatureAddition, - RecursiveFeatureElimination, -) +from feature_engine.selection import RecursiveFeatureElimination _selectors = [ RecursiveFeatureElimination, - RecursiveFeatureAddition, ] _input_params = [ @@ -158,12 +154,6 @@ def test_fit_initial_model_performance( def test_feature_importances(_estimator, _importance, df_test): X, y = df_test - sel = RecursiveFeatureAddition(_estimator, threshold=-100).fit(X, y) - _importance.sort(reverse=True) - assert ( - list(np.round(sel.feature_importances_.values, 2)) == np.round(_importance, 2) - ).all() - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) _importance.sort(reverse=False) assert ( @@ -201,11 +191,6 @@ def test_permutation_importance(_estimator, df_test): expected_importances.index = X.columns expected_importances = expected_importances.mean(axis=1) - sel = RecursiveFeatureAddition(_estimator, threshold=-100).fit(X, y) - assert_series_equal( - sel.feature_importances_.sort_index(), expected_importances.sort_index() - ) - sel = RecursiveFeatureElimination(_estimator, threshold=-100).fit(X, y) assert_series_equal( sel.feature_importances_.sort_index(), expected_importances.sort_index() @@ -215,10 +200,6 @@ def test_permutation_importance(_estimator, df_test): @pytest.mark.parametrize("_estimator", ests) def test_selection_after_permutation_importance(_estimator, df_test): X, y = df_test - sel = RecursiveFeatureAddition(_estimator) - Xtr = sel.fit_transform(X, y) - assert Xtr.shape[1] < X.shape[1] - sel = RecursiveFeatureElimination(_estimator) Xtr = sel.fit_transform(X, y) assert Xtr.shape[1] < X.shape[1]