diff --git a/docs/user_guide/selection/SelectByShuffling.rst b/docs/user_guide/selection/SelectByShuffling.rst index 61043945d..b144c274d 100644 --- a/docs/user_guide/selection/SelectByShuffling.rst +++ b/docs/user_guide/selection/SelectByShuffling.rst @@ -104,7 +104,7 @@ the entire dataset, without shuffling, using cross-validation. .. code:: python - 0.488702767247119 + 0.48870212980353145 In the following sections, we'll explore some of the additional useful data stored by :class:`SelectByShuffling()`. @@ -125,15 +125,15 @@ each feature: .. code:: python {'age': -0.0054698043007869734, - 'sex': 0.03325633986510784, - 'bmi': 0.184158237207512, + 'sex': 0.03325633986510779, + 'bmi': 0.1841582372075119, 'bp': 0.10089894421748086, - 's1': 0.49324432634948095, - 's2': 0.21163252880660438, - 's3': 0.02006839198785859, - 's4': 0.011098050006761673, - 's5': 0.4828781996541602, - 's6': 0.003963360084439538} + 's1': 0.4932443263494815, + 's2': 0.2116325288066046, + 's3': 0.020068391987858536, + 's4': 0.011098050006761617, + 's5': 0.4828781996541603, + 's6': 0.003963360084439482} :class:`SelectByShuffling()` stores the standard deviation of the performance change: @@ -146,16 +146,16 @@ shuffling: .. code:: python - {'age': 0.012788500580799392, + {'age': 0.012788500580799462, 'sex': 0.040792331972680645, - 'bmi': 0.042212436355346106, - 'bp': 0.05397012536801143, - 's1': 0.35198797776358015, - 's2': 0.167636042355086, - 's3': 0.03455158514716544, - 's4': 0.007755675852874145, - 's5': 0.1449579162698361, - 's6': 0.011193022434166025} + 'bmi': 0.04221243635534631, + 'bp': 0.053970125368011726, + 's1': 0.3519879777635807, + 's2': 0.1676360423550863, + 's3': 0.03455158514716549, + 's4': 0.007755675852874141, + 's5': 0.14495791626983628, + 's6': 0.011193022434166125} We can plot the performance change together with the standard deviation to get a better idea of how shuffling features affect the model performance: @@ -181,8 +181,8 @@ feature: .. figure:: ../../images/shuffle-features-std.png -With this set up, features that elicited a mean performance drop greater than the mean -performance of all features, will be removed. If, for any reason, this threshold is too +With this set up, features that elicited a performance drop smaller than the mean +performance drop of all features will be removed. If, for any reason, this threshold is too conservative or too permissive, by analysing the former barplot, you can get a better idea of how these features affect the predictions of the model, and select a different threshold. @@ -198,7 +198,7 @@ threshold: tr.features_to_drop_ The following features were deemed as non-important, because their performance drift is -greater than the mean performance drift of all features: +smaller than the mean performance drift of all features: .. code:: python @@ -221,7 +221,52 @@ In the following output, we see the dataframe with the selected features: 3 -0.011595 0.012191 0.024991 0.022688 4 -0.036385 0.003935 0.015596 -0.031988 - +Using polars +~~~~~~~~~~~~ + +:class:`SelectByShuffling()` also works with polars dataframes, and returns a polars +dataframe when the input is a polars dataframe. With the same `random_state`, the +features are shuffled in the same way as with pandas, so the performance drifts and the +selected features are the same. + +.. code:: python + + import polars as pl + + X_pl = pl.DataFrame(X.to_dict(orient="list")) + y_pl = pl.Series("target", y.to_list()) + + tr = SelectByShuffling( + estimator=LinearRegression(), + scoring="r2", + cv=3, + random_state=0, + ) + + Xt = tr.fit_transform(X_pl, y_pl) + + print(tr.features_to_drop_) + print(Xt.head()) + +In the following output, we see the features to drop and the polars dataframe with the +selected features: + +.. code:: text + + ['age', 'sex', 'bp', 's3', 's4', 's6'] + shape: (5, 4) + ┌───────────┬───────────┬───────────┬───────────┐ + │ bmi ┆ s1 ┆ s2 ┆ s5 │ + │ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞═══════════╪═══════════╪═══════════╪═══════════╡ + │ 0.061696 ┆ -0.044223 ┆ -0.034821 ┆ 0.019907 │ + │ -0.051474 ┆ -0.008449 ┆ -0.019163 ┆ -0.068332 │ + │ 0.044451 ┆ -0.045599 ┆ -0.034194 ┆ 0.002861 │ + │ -0.011595 ┆ 0.012191 ┆ 0.024991 ┆ 0.022688 │ + │ -0.036385 ┆ 0.003935 ┆ 0.015596 ┆ -0.031988 │ + └───────────┴───────────┴───────────┴───────────┘ + Additional resources -------------------- diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/selection/shuffle_features.py b/feature_engine/selection/shuffle_features.py index 946b963d3..a3d66273b 100644 --- a/feature_engine/selection/shuffle_features.py +++ b/feature_engine/selection/shuffle_features.py @@ -1,11 +1,13 @@ from types import GeneratorType from typing import List, MutableSequence, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd -from sklearn.base import is_classifier +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.metrics import get_scorer -from sklearn.model_selection import check_cv, cross_validate +from sklearn.model_selection import cross_validate +from sklearn.utils import _safe_indexing from sklearn.utils.validation import _check_sample_weight, check_random_state from feature_engine._check_init_parameters.check_variables import ( @@ -169,6 +171,37 @@ class SelectByShuffling(BaseSelector): 3 1 1 1 4 2 0 1 5 2 1 1 + + With polars: + + >>> import polars as pl + >>> from sklearn.ensemble import RandomForestClassifier + >>> from feature_engine.selection import SelectByShuffling + >>> X = pl.DataFrame(dict(x1 = [1000,2000,1000,1000,2000,3000], + >>> x2 = [2,4,3,1,2,2], + >>> x3 = [1,1,1,0,0,0], + >>> x4 = [1,2,1,1,0,1], + >>> x5 = [1,1,1,1,1,1])) + >>> y = pl.Series([1,0,0,1,1,0]) + >>> sbs = SelectByShuffling( + >>> RandomForestClassifier(random_state=42), + >>> cv=2, + >>> random_state=42, + >>> ) + >>> sbs.fit_transform(X, y) + shape: (6, 3) + ┌─────┬─────┬─────┐ + │ x2 ┆ x4 ┆ x5 │ + │ --- ┆ --- ┆ --- │ + │ i64 ┆ i64 ┆ i64 │ + ╞═════╪═════╪═════╡ + │ 2 ┆ 1 ┆ 1 │ + │ 4 ┆ 2 ┆ 1 │ + │ 3 ┆ 1 ┆ 1 │ + │ 1 ┆ 1 ┆ 1 │ + │ 2 ┆ 0 ┆ 1 │ + │ 2 ┆ 1 ┆ 1 │ + └─────┴─────┴─────┘ """ def __init__( @@ -182,8 +215,11 @@ def __init__( confirm_variables: bool = False, ): - if threshold and not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float or None") + if threshold is not None and not isinstance(threshold, (int, float)): + raise ValueError( + "threshold must be an integer, a float or None. " + f"Got {threshold} instead." + ) super().__init__(confirm_variables) @@ -196,8 +232,8 @@ def __init__( def fit( self, - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y: IntoSeries, sample_weight: Union[MutableSequence, None] = None, ): """ @@ -205,8 +241,9 @@ def fit( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The input dataframe. + X: dataframe of shape = [n_samples, n_features] + The input dataframe. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: array-like of shape (n_samples) Target variable. Required to train the estimator. @@ -214,12 +251,7 @@ def fit( sample_weight : array-like of shape (n_samples,), default=None Sample weights. If None, then samples are equally weighted. """ - - X, y = check_X_y(X, y) - - # reset the index - X = X.reset_index(drop=True) - y = y.reset_index(drop=True) + nw_X, y = check_X_y(X, y) if sample_weight is not None: sample_weight = _check_sample_weight(sample_weight, X) @@ -233,70 +265,76 @@ def fit( cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv + nw_X_model = nw_X.select(nw.col(*self.variables_)) + X_model = nw_X_model.to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=cv, return_estimator=True, + return_indices=True, scoring=self.scoring, params={"sample_weight": sample_weight}, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # extract the validation folds - cv_ = check_cv(cv, y=y, classifier=is_classifier(self.estimator)) - validation_indices = [val_index for _, val_index in cv_.split(X, y)] + # the indices returned by cross_validate are the folds each model was + # evaluated on, also when the splitter gives different folds on every call. + validation_indices = model["indices"]["test"] + y_val = [_safe_indexing(y, idx) for idx in validation_indices] - # get performance metric scorer = get_scorer(self.scoring) - - # seed random_state = check_random_state(self.random_state) + n_samples = nw_X.shape[0] - # dict to collect features and their performance_drift after shuffling self.performance_drifts_ = {} self.performance_drifts_std_ = {} - # shuffle features and save feature performance drift into a dict - for feature in self.variables_: - - X_shuffled = X[self.variables_].copy() + estimators = model["estimator"] - # shuffle individual feature - X_shuffled[feature] = ( - X_shuffled[feature] - .sample(frac=1, random_state=random_state) - .reset_index(drop=True) - ) - - # determine the performance with the shuffled feature - performance = [ - scorer(m, X_shuffled.iloc[idx], y.iloc[idx]) - for m, idx in zip(model["estimator"], validation_indices) - ] - - performance_std = np.std(performance) - performance = np.mean(performance) - - # determine drift in performance - # Note, sklearn negates the log and error scores, so no need to manually - # do the inversion - # https://scikit-learn.org/stable/modules/model_evaluation.html - # (https://scikit-learn.org/stable/modules/model_evaluation.html - # #the-scoring-parameter-defining-model-evaluation-rules) - performance_drift = self.initial_model_performance_ - performance + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + X_val = [X_model.iloc[idx] for idx in validation_indices] + else: + nw_X_val = [nw_X_model[idx] for idx in validation_indices] - # Save feature and performance drift - self.performance_drifts_[feature] = performance_drift - self.performance_drifts_std_[feature] = performance_std + for feature in self.variables_: + # same permutation as pandas' sample(frac=1), so the drifts for a given + # random_state are the same with every dataframe library. + permutation = random_state.permutation(n_samples) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + values = X_model[feature].array + performance = [] + for m, idx, X_, y_ in zip(estimators, validation_indices, X_val, y_val): + # the fold dataframes are copies made in fit, so we can shuffle + # the column in place and restore it afterwards. + original = X_[feature] + X_[feature] = values.take(permutation[idx]) + performance.append(scorer(m, X_, y_)) + X_[feature] = original + else: + column = nw_X_model.get_column(feature) + performance = [ + scorer(m, X_.with_columns(column[permutation[idx]]).to_native(), y_) + for m, idx, X_, y_ in zip( + estimators, validation_indices, nw_X_val, y_val + ) + ] + + # sklearn negates the error and loss scores, so larger is always better. + drift = self.initial_model_performance_ - np.mean(performance) + self.performance_drifts_[feature] = drift + self.performance_drifts_std_[feature] = np.std(performance) # select features if not self.threshold: - threshold = pd.Series(self.performance_drifts_).mean() + threshold = np.mean(list(self.performance_drifts_.values())) else: threshold = self.threshold @@ -306,7 +344,6 @@ def fit( if self.performance_drifts_[f] < threshold ] - # save input features self._get_feature_names_in(X) return self diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) diff --git a/tests/test_selection/test_shuffle_features.py b/tests/test_selection/test_shuffle_features.py index 5f1e65dbb..badfbdaa9 100644 --- a/tests/test_selection/test_shuffle_features.py +++ b/tests/test_selection/test_shuffle_features.py @@ -1,192 +1,392 @@ +import re + import numpy as np import pandas as pd import pytest - -from sklearn.ensemble import RandomForestClassifier -from sklearn.linear_model import LinearRegression -from sklearn.model_selection import StratifiedKFold +from sklearn.datasets import load_diabetes +from sklearn.ensemble import HistGradientBoostingClassifier, RandomForestClassifier +from sklearn.linear_model import LinearRegression, LogisticRegression +from sklearn.model_selection import KFold, StratifiedKFold from sklearn.tree import DecisionTreeRegressor from feature_engine.selection import SelectByShuffling +from tests.backend_helpers import frame_to_dict, make_series + +VARIABLES = ["var_0", "var_4", "var_7", "var_9"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _random_forest(): + return RandomForestClassifier(n_estimators=10, random_state=1) + + +@pytest.fixture(scope="module") +def data_diabetes(): + X, y = load_diabetes(return_X_y=True, as_frame=True) + data = {col: X[col].tolist() for col in X.columns} + data["target"] = y.tolist() + return data -def test_sel_with_default_parameters(df_test): - X, y = df_test +# init parameters +@pytest.mark.parametrize("threshold", ["hello", "", [0.1], [], {"a": 1}, (1,)]) +def test_error_if_threshold_not_number_or_none(threshold): + msg = f"threshold must be an integer, a float or None. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + SelectByShuffling(_random_forest(), threshold=threshold) + + +@pytest.mark.parametrize("confirm_variables", ["True", 1, None, [True]]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + SelectByShuffling(_random_forest(), confirm_variables=confirm_variables) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, threshold, random_state, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, None, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), 0.01, 1, True), + (LinearRegression(), "r2", KFold(2), 5, 10, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, threshold, random_state, confirm_variables +): sel = SelectByShuffling( - RandomForestClassifier(random_state=1), threshold=0.01, random_state=1 + estimator, + scoring=scoring, + cv=cv, + threshold=threshold, + random_state=random_state, + confirm_variables=confirm_variables, + ) + assert sel.estimator is estimator + assert sel.scoring == scoring + assert sel.cv is cv + assert sel.threshold == threshold + assert sel.random_state == random_state + assert sel.confirm_variables is confirm_variables + + +# fit and transform +def test_fit_attributes(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 ) sel.fit(X, y) - # expected result - Xtransformed = pd.DataFrame(X["var_7"].copy()) + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.0037026895460631204, + "var_4": 0.0017710452198405058, + "var_7": 0.25884577270119447, + "var_9": -1.959497441428315e-05, + } + ) + assert sel.performance_drifts_std_ == pytest.approx( + { + "var_0": 0.0005724303014393721, + "var_4": 0.0005148668118348698, + "var_7": 0.02708631411106102, + "var_9": 0.0016362909674612345, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + assert sel.variables_ == VARIABLES + assert sel.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert sel.n_features_in_ == 12 + + +def test_transform_removes_features_below_threshold(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = SelectByShuffling(_random_forest(), threshold=0.01, random_state=1) + Xt = sel.fit_transform(X, y) - # test init params - assert sel.threshold == 0.01 - assert sel.cv == 3 - assert sel.scoring == "roc_auc" - # test fit attrs - assert np.round(sel.initial_model_performance_, 3) == 0.997 + assert sel.initial_model_performance_ == pytest.approx(0.9953593232960704) assert sel.features_to_drop_ == [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", + f"var_{i}" for i in [0, 1, 2, 3, 4, 5, 6, 8, 9, 10, 11] ] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {"var_7": data_classification["var_7"]} -def test_regression_cv_3_and_r2(load_diabetes_dataset): - # test for regression using cv=3, and the r2 as metric. - X, y = load_diabetes_dataset - +def test_threshold_none_uses_mean_drift(make_df, data_classification): + X, y = _split_target(make_df, data_classification) sel = SelectByShuffling( - estimator=LinearRegression(), scoring="r2", cv=3, threshold=0.05, random_state=1 + LogisticRegression(), scoring="accuracy", variables=VARIABLES, random_state=3 ) sel.fit(X, y) - # expected output - Xtransformed = X[[1, 2, 3, 4, 5, 8]].copy() + # the mean drift is 0.13, only var_7 is above it. + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.021998045950141765, + "var_4": 0.005002007996020019, + "var_7": 0.4890009770249292, + "var_9": 0.004001006995019041, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] - # test init params - assert sel.cv == 3 - assert sel.scoring == "r2" - assert sel.threshold == 0.05 - # fit params - assert np.round(sel.initial_model_performance_, 3) == 0.489 - assert sel.features_to_drop_ == [0, 6, 7, 9] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +def test_regression_with_r2(make_df, data_diabetes): + X, y = _split_target(make_df, data_diabetes) + sel = SelectByShuffling( + LinearRegression(), scoring="r2", cv=3, threshold=0.05, random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.48870212980353145) + assert sel.performance_drifts_ == pytest.approx( + { + "age": 0.0019221800216577822, + "sex": 0.058082587710752975, + "bmi": 0.16770566452186242, + "bp": 0.0702676565845552, + "s1": 0.5151999913097192, + "s2": 0.17198756874519755, + "s3": 0.01669668794929896, + "s4": 0.025518182049075577, + "s5": 0.4065911011104548, + "s6": 0.0038028879119369474, + } + ) + assert sel.features_to_drop_ == ["age", "s3", "s4", "s6"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["sex", "bmi", "bp", "s1", "s2", "s5"] -def test_regression_cv_2_and_mse(load_diabetes_dataset): - # test for regression using cv=2, and the neg_mean_squared_error as metric. - # add suitable threshold for regression mse - X, y = load_diabetes_dataset +def test_regression_with_neg_mse_and_target_as_dataframe(make_df, data_diabetes): + X, _ = _split_target(make_df, data_diabetes) + y = make_df({"target": data_diabetes["target"]}) sel = SelectByShuffling( - estimator=DecisionTreeRegressor(random_state=0), + DecisionTreeRegressor(random_state=0), scoring="neg_mean_squared_error", cv=2, threshold=1000, random_state=1, ) - # fit transformer sel.fit(X, y) - # expected output - Xtransformed = X[[2, 7, 8]].copy() + assert sel.initial_model_performance_ == pytest.approx(-5835.58371040724) + assert sel.performance_drifts_ == pytest.approx( + { + "age": -227.27149321266916, + "sex": -78.94570135746653, + "bmi": 2132.739819004524, + "bp": 134.5814479638011, + "s1": 279.30316742081413, + "s2": 313.13800904977325, + "s3": 19.719457013574356, + "s4": 2050.5294117647063, + "s5": 1661.9977375565613, + "s6": -4.617647058823422, + } + ) + assert sel.features_to_drop_ == ["age", "sex", "bp", "s1", "s2", "s3", "s6"] + + +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_cv_options(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, cv=cv, threshold=0.01, random_state=1 + ) + sel.fit(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + - # test init params - assert sel.cv == 2 - assert sel.scoring == "neg_mean_squared_error" - assert sel.threshold == 1000 - # fit params - assert np.round(sel.initial_model_performance_, 0) == -5836.0 - assert sel.features_to_drop_ == [0, 1, 3, 4, 5, 6, 9] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +class _FoldsChangeOnEachCall(KFold): + """Splitter that yields different folds every time split() is called.""" + def split(self, X, y=None, groups=None): + self.n_calls = getattr(self, "n_calls", 0) + 1 + folds = list(super().split(X, y, groups)) + return iter(folds if self.n_calls % 2 == 1 else folds[::-1]) -def test_cv_generator(df_test): - X, y = df_test - cv = StratifiedKFold(n_splits=3) - X, y = df_test +def test_performance_is_evaluated_on_the_folds_of_each_model( + make_df, data_classification +): + # the shuffled data must be scored on the fold each model was not trained on. + X, y = _split_target(make_df, data_classification) + params = dict(variables=VARIABLES, random_state=1) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, - random_state=1, - cv=3, + _random_forest(), cv=_FoldsChangeOnEachCall(n_splits=3), **params ) sel.fit(X, y) + reference = SelectByShuffling(_random_forest(), cv=KFold(n_splits=3), **params) + reference.fit(X, y) + + assert sel.performance_drifts_ == pytest.approx(reference.performance_drifts_) + assert sel.performance_drifts_std_ == pytest.approx( + reference.performance_drifts_std_ + ) + + +def test_non_numerical_variables_are_ignored(make_df, data_classification): + data = {**data_classification, "cat_1": ["a"] * 1000, "cat_2": ["b"] * 1000} + X, y = _split_target(make_df, data) + sel = SelectByShuffling(_random_forest(), threshold=0.01, random_state=1) + Xt = sel.fit_transform(X, y) - # expected result - Xtransformed = pd.DataFrame(X["var_7"].copy()) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert sel.variables_ == [f"var_{i}" for i in range(12)] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_7": data_classification["var_7"], + "cat_1": ["a"] * 1000, + "cat_2": ["b"] * 1000, + } + +def test_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, + _random_forest(), + variables=["var_0", "var_4", "var_7", "not_in_X"], + confirm_variables=True, random_state=1, - cv=cv, ) sel.fit(X, y) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + assert sel.variables_ == ["var_0", "var_4", "var_7"] + assert sel.features_to_drop_ == ["var_0", "var_4"] + + +def test_missing_values(make_df, data_classification): + data = {k: data_classification[k] for k in [*VARIABLES, "target"]} + data["var_0"] = [None if i % 7 == 0 else v for i, v in enumerate(data["var_0"])] + data["var_7"] = [None if i % 5 == 0 else v for i, v in enumerate(data["var_7"])] + X, y = _split_target(make_df, data) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), - threshold=0.01, - random_state=1, - cv=cv.split(X, y), + HistGradientBoostingClassifier(max_iter=10, random_state=0), random_state=1 ) sel.fit(X, y) - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) + + assert sel.initial_model_performance_ == pytest.approx(0.993134587411696) + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.008290557612846916, + "var_4": 0.35213096252252896, + "var_7": 0.002123827199128625, + "var_9": 0.000764131562324577, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_7", "var_9"] -def test_raises_threshold_error(): - with pytest.raises(ValueError): - SelectByShuffling(RandomForestClassifier(random_state=1), threshold="hello") +def test_sample_weight(make_df): + X = make_df( + { + "x1": [1000, 2000, 1000, 1000, 2000, 3000], + "x2": [1000, 2000, 1000, 1000, 2000, 3000], + } + ) + y = make_series(make_df, [1, 0, 0, 1, 1, 0]) + sel = SelectByShuffling( + RandomForestClassifier(random_state=42), cv=2, random_state=42 + ) + sel.fit(X, y, sample_weight=[1000, 2000, 1000, 1000, 2000, 3000]) + assert sel.initial_model_performance_ == 0.125 + assert sel.features_to_drop_ == ["x2"] -def test_automatic_variable_selection(df_test): - X, y = df_test - # add 2 additional categorical variables, these should not be evaluated by - # the selector - X["cat_1"] = "cat1" - X["cat_2"] = "cat2" +@pytest.mark.parametrize("to_target", [list, np.array]) +def test_target_as_list_and_array(make_df, data_classification, to_target): + X, _ = _split_target(make_df, data_classification) + y = to_target(data_classification["target"]) sel = SelectByShuffling( - RandomForestClassifier(random_state=1), threshold=0.01, random_state=1 + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 ) sel.fit(X, y) - # expected result - Xtransformed = X[["var_7", "cat_1", "cat_2"]].copy() + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] - # test init params - assert sel.threshold == 0.01 - assert sel.cv == 3 - assert sel.scoring == "roc_auc" - # test fit attrs - assert np.round(sel.initial_model_performance_, 3) == 0.997 - assert sel.features_to_drop_ == [ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_8", - "var_9", - "var_10", - "var_11", - ] - # test transform output - pd.testing.assert_frame_equal(sel.transform(X), Xtransformed) +def test_input_dataframe_is_not_modified(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + SelectByShuffling(_random_forest(), variables=VARIABLES, random_state=1).fit(X, y) -def test_sample_weights(): - X = pd.DataFrame( - dict( - x1=[1000, 2000, 1000, 1000, 2000, 3000], - x2=[1000, 2000, 1000, 1000, 2000, 3000], - ) + assert frame_to_dict(X) == { + k: v for k, v in data_classification.items() if k != "target" + } + + +def test_integer_column_names_pandas(data_classification): + X = pd.DataFrame({i: data_classification[f"var_{i}"] for i in range(12)}) + y = pd.Series(data_classification["target"]) + sel = SelectByShuffling( + _random_forest(), variables=[0, 4, 7, 9], threshold=0.01, random_state=1 ) - y = pd.Series([1, 0, 0, 1, 1, 0]) + Xt = sel.fit_transform(X, y) + + assert sel.performance_drifts_ == pytest.approx( + { + 0: 0.0037026895460631204, + 4: 0.0017710452198405058, + 7: 0.25884577270119447, + 9: -1.959497441428315e-05, + } + ) + assert sel.features_to_drop_ == [0, 4, 9] + pd.testing.assert_frame_equal(Xt, X.drop(columns=[0, 4, 9])) - sbs = SelectByShuffling( - RandomForestClassifier(random_state=42), cv=2, random_state=42 + +def test_non_default_index_pandas(data_classification): + index = np.arange(1000)[::-1] + 50 + X = pd.DataFrame( + {k: v for k, v in data_classification.items() if k != "target"}, index=index ) + y = pd.Series(data_classification["target"], index=index) + sel = SelectByShuffling( + _random_forest(), variables=VARIABLES, threshold=0.01, random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.initial_model_performance_ == pytest.approx(0.9965735235313549) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + pd.testing.assert_frame_equal(Xt, X.drop(columns=["var_0", "var_4", "var_9"])) + - sample_weight = [1000, 2000, 1000, 1000, 2000, 3000] - sbs.fit_transform(X, y, sample_weight=sample_weight) - assert sbs.initial_model_performance_ == 0.125 +def test_nullable_integer_dtype_pandas(data_classification): + X = pd.DataFrame({k: data_classification[k] for k in VARIABLES}) + X["var_0"] = pd.array( + [None if i % 7 == 0 else round(v * 10) for i, v in enumerate(X["var_0"])], + dtype="Int64", + ) + y = pd.Series(data_classification["target"]) + sel = SelectByShuffling( + HistGradientBoostingClassifier(max_iter=10, random_state=0), random_state=1 + ) + Xt = sel.fit_transform(X, y) + + assert sel.variables_ == VARIABLES + assert sel.performance_drifts_ == pytest.approx( + { + "var_0": 0.0007341414720931638, + "var_4": -0.00020804719599920585, + "var_7": 0.45245947715827217, + "var_9": 0.0001250311491274303, + } + ) + assert sel.features_to_drop_ == ["var_0", "var_4", "var_9"] + pd.testing.assert_frame_equal(Xt, X[["var_7"]])