diff --git a/docs/user_guide/selection/SmartCorrelatedSelection.rst b/docs/user_guide/selection/SmartCorrelatedSelection.rst index ecf13b169..8a8205ba2 100644 --- a/docs/user_guide/selection/SmartCorrelatedSelection.rst +++ b/docs/user_guide/selection/SmartCorrelatedSelection.rst @@ -44,8 +44,8 @@ feature on the target. Procedure --------- -:class:`SmartCorrelatedSelection` first finds correlated feature groups using any -correlation method supported by `pandas.corr()`, or a user defined function that returns +:class:`SmartCorrelatedSelection` first finds correlated feature groups using the Pearson, +Spearman or Kendall correlation coefficient, or a user defined function that returns a value between -1 and 1. Then, from each group of correlated features, it will try and identify the best candidate @@ -561,6 +561,57 @@ for those that are dropped: [True, True, True, True, False, True, False, True, False, False, True, True] +With polars +~~~~~~~~~~~ + +:class:`SmartCorrelatedSelection` also works with polars dataframes, and returns a polars +dataframe. Let's repeat the selection based on the correlation with the target, with the +data from the previous example as a polars dataframe and series: + +.. code:: python + + import polars as pl + + X_pl = pl.DataFrame(X.to_dict(orient="list")) + y_pl = pl.Series(y.tolist()) + + tr = SmartCorrelatedSelection(threshold=0.8, selection_method="corr_with_target") + Xt = tr.fit_transform(X_pl, y_pl) + +The selector drops the same features as with pandas: + +.. code:: python + + tr.features_to_drop_ + +.. code:: python + + ['var_4', 'var_6', 'var_9', 'var_8'] + +And returns a polars dataframe: + +.. code:: python + + print(Xt.head()) + +.. code:: text + + shape: (5, 8) + ┌──────────┬──────────┬───────────┬───────────┬───────────┬───────────┬───────────┬───────────┐ + │ var_0 ┆ var_1 ┆ var_2 ┆ var_3 ┆ var_5 ┆ var_7 ┆ var_10 ┆ var_11 │ + │ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- ┆ --- │ + │ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 ┆ f64 │ + ╞══════════╪══════════╪═══════════╪═══════════╪═══════════╪═══════════╪═══════════╪═══════════╡ + │ 1.471061 ┆ -2.3764 ┆ -0.247208 ┆ 1.21029 ┆ 0.091527 ┆ -2.23017 ┆ 2.070526 ┆ -1.989335 │ + │ 1.819196 ┆ 1.969326 ┆ -0.126894 ┆ 0.034598 ┆ -0.186802 ┆ -1.44749 ┆ 1.18482 ┆ -1.309524 │ + │ 1.625024 ┆ 1.499174 ┆ 0.334123 ┆ -2.233844 ┆ -0.313881 ┆ -2.240741 ┆ -0.066448 ┆ -0.852703 │ + │ 1.939212 ┆ 0.075341 ┆ 1.627132 ┆ 0.943132 ┆ -0.468041 ┆ -3.534861 ┆ 0.713558 ┆ 0.484649 │ + │ 1.579307 ┆ 0.372213 ┆ 0.338141 ┆ 0.951526 ┆ 0.729005 ┆ -2.053965 ┆ 0.39879 ┆ -0.18653 │ + └──────────┴──────────┴───────────┴───────────┴───────────┴───────────┴───────────┴───────────┘ + +With polars, the missing data are the null values. When `selection_method="missing_values"`, +the selector counts the nulls in each feature, and when `selection_method="cardinality"`, it +does not count null as a value. And that's it! diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/selection/smart_correlation_selection.py b/feature_engine/selection/smart_correlation_selection.py index ec58e2da4..1e471a788 100644 --- a/feature_engine/selection/smart_correlation_selection.py +++ b/feature_engine/selection/smart_correlation_selection.py @@ -1,7 +1,11 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries +from scipy.stats import kendalltau, spearmanr from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -29,8 +33,9 @@ _check_contains_inf, _check_contains_na, check_X, - check_y, + check_X_y, ) +from feature_engine.encoding._helper_functions import TARGET_NAME, add_target_to_X from feature_engine.selection.base_selector import BaseSelector from .base_selection_functions import ( @@ -71,8 +76,6 @@ class SmartCorrelatedSelection(BaseSelector): correlated features, the selected variable, plus all the features that were not correlated to any other. - Correlation is calculated with `pandas.corr()`. - SmartCorrelatedSelection() works only with numerical variables. Categorical variables will need to be encoded to numerical or will be excluded from the analysis. @@ -92,8 +95,6 @@ class SmartCorrelatedSelection(BaseSelector): - 'spearman': Spearman rank correlation - callable: callable with input two 1d ndarrays and returning a float. - For more details on this parameter visit the `pandas.corr()` documentation. - threshold: float, default=0.8 The correlation threshold above which a feature will be deemed correlated with another one and removed from the dataset. @@ -173,7 +174,6 @@ class SmartCorrelatedSelection(BaseSelector): See Also -------- - pandas.corr feature_engine.selection.DropCorrelatedFeatures Examples @@ -186,10 +186,10 @@ class SmartCorrelatedSelection(BaseSelector): >>> x3 = [1, 0, 0, 0])) >>> scs = SmartCorrelatedSelection(threshold=0.7) >>> scs.fit_transform(X) - x2 x3 - 0 2 1 - 1 4 0 - 2 3 0 + x1 x3 + 0 1 1 + 1 2 0 + 2 1 0 3 1 0 It is also possible to use alternative selection methods. Here, we select those @@ -205,6 +205,27 @@ class SmartCorrelatedSelection(BaseSelector): 1 2000 0 2 1500 0 3 500 0 + + With polars: + + >>> import polars as pl + >>> from feature_engine.selection import SmartCorrelatedSelection + >>> X = pl.DataFrame(dict(x1 = [2,4,3,1], + ... x2 = [1000,2000,1500,500], + ... x3 = [1, 0, 0, 0])) + >>> scs = SmartCorrelatedSelection(threshold=0.7, selection_method="variance") + >>> scs.fit_transform(X) + shape: (4, 2) + ┌──────┬─────┐ + │ x2 ┆ x3 │ + │ --- ┆ --- │ + │ i64 ┆ i64 │ + ╞══════╪═════╡ + │ 1000 ┆ 1 │ + │ 2000 ┆ 0 │ + │ 1500 ┆ 0 │ + │ 500 ┆ 0 │ + └──────┴─────┘ """ def __init__( @@ -225,13 +246,16 @@ def __init__( f"`threshold` must be a float between 0 and 1. Got {threshold} instead." ) - if missing_values not in ["raise", "ignore"]: + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + ]: raise ValueError( "missing_values takes only values 'raise' or 'ignore'. " f"Got {missing_values} instead." ) - if selection_method not in [ + if not isinstance(selection_method, str) or selection_method not in [ "missing_values", "cardinality", "variance", @@ -269,23 +293,30 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Find the correlated feature groups. Determine which feature should be selected from each group. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. - y: pandas series. Default = None + y: Series, list or numpy array of shape = [n_samples], default=None y is needed if selection_method == 'model_performance' or 'corr_with_target'. """ - # check input dataframe - X = check_X(X) + if self.selection_method in ["model_performance", "corr_with_target"]: + if y is None: + raise ValueError( + f"When `selection_method = '{self.selection_method}'` y is " + "needed to fit the transformer." + ) + nw_X, y = check_X_y(X, y) + else: + nw_X = check_X(X) self.variables_ = _select_numerical_variables( X, self.variables, self.confirm_variables @@ -299,47 +330,37 @@ def fit(self, X: pd.DataFrame, y: pd.Series = None): _check_contains_na(X, self.variables_) _check_contains_inf(X, self.variables_) - if ( - self.selection_method in ["model_performance", "corr_with_target"] - ) and y is None: - raise ValueError( - f"When `selection_method = '{self.selection_method}'` y is needed to " - "fit the transformer." - ) - - if self.selection_method == "missing_values": - features = ( - X[self.variables_] - .isnull() - .sum() - .sort_values(ascending=True, kind="mergesort") - .index.to_list() - ) - elif self.selection_method == "variance": - features = ( - X[self.variables_] - .std() - .sort_values(ascending=False, kind="mergesort") - .index.to_list() - ) - elif self.selection_method == "cardinality": - features = ( - X[self.variables_] - .nunique() - .sort_values(ascending=False, kind="mergesort") - .index.to_list() - ) + if self.selection_method == "model_performance": + features = sorted(self.variables_) elif self.selection_method == "corr_with_target": - y = check_y(y) - features = ( - X[self.variables_] - .corrwith(y, method=self.method) - .abs() - .sort_values(ascending=False, kind="mergesort") - .index.to_list() + correlation = _correlation_with_target( + X, nw_X, self.variables_, y, self.method ) + features = _sort_features(self.variables_, np.abs(correlation), False) + elif self.selection_method == "missing_values": + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + missing = X[self.variables_].isna().sum().to_numpy() + else: + missing = nw_X.select(nw.col(*self.variables_).null_count()).row(0) + features = _sort_features(self.variables_, missing, True) + elif self.selection_method == "variance": + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + std = X[self.variables_].std().to_numpy() + else: + std = nw_X.select(nw.col(*self.variables_).std()).row(0) + features = _sort_features(self.variables_, std, False) else: - features = sorted(self.variables_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + cardinality = X[self.variables_].nunique().to_numpy() + else: + # like pandas, nulls are not counted as a value. + cardinality = nw_X.select( + nw.col(*self.variables_).drop_nulls().n_unique() + ).row(0) + features = _sort_features(self.variables_, cardinality, False) correlated_groups, features_to_drop, correlated_dict = find_correlated_features( X, features, self.method, self.threshold @@ -353,18 +374,18 @@ def fit(self, X: pd.DataFrame, y: pd.Series = None): feature_performance, _ = single_feature_performance( X=X, y=y, - variables=feature_group, + variables=sorted(feature_group), estimator=self.estimator, cv=cv, groups=self.groups, scoring=self.scoring, ) # get most important feature - f_i = ( - pd.Series(feature_performance) - .sort_values(ascending=False, kind="mergesort") - .index[0] - ) + f_i = _sort_features( + list(feature_performance), + list(feature_performance.values()), + False, + )[0] correlated_dict[f_i] = feature_group.difference({f_i}) # convoluted way to pick up the variables from the sets in the @@ -383,3 +404,75 @@ def fit(self, X: pd.DataFrame, y: pd.Series = None): self._get_feature_names_in(X) return self + + +def _sort_features(features: list, values, ascending: bool) -> list: + """ + Sort the features by their values. Like pandas' stable sort, ties keep the order + of the features and NaN go last. + """ + values = np.asarray(values, dtype=float) + order = np.argsort(values if ascending is True else -values, kind="stable") + return [features[i] for i in order] + + +def _correlation_with_target( + X: IntoDataFrame, nw_X: nw.DataFrame, variables: list, y, method +) -> np.ndarray: + """ + Correlation of each variable with the target. Like pandas.DataFrame.corrwith(), + each variable is compared with the target on the rows where it is not NaN. + """ + # polars is faster than numpy and narwhals. + if method in ["pearson", "spearman"] and nwd.is_polars_dataframe(X) is True: + native_ns = nw.get_native_namespace(X) + target = native_ns.col(TARGET_NAME) + exprs = [] + for var in variables: + # null and NaN are missing values, like in pandas; inf are not. + rows = native_ns.col(var).is_not_nan() + corr = native_ns.corr( + native_ns.col(var).filter(rows), target.filter(rows), method=method + ) + exprs.append(corr.alias(var)) + Xy = add_target_to_X(nw_X, y).to_native() + return np.array(Xy.select(exprs).row(0), dtype=float) + + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + values = nw_X.select(nw.col(*variables)).to_numpy().astype(float) + target = np.asarray(y, dtype=float) + + # numpy and scipy are faster than pandas, and return the same values. + if method == "pearson": + corr_func = _pearson + elif method == "spearman": + corr_func = _spearman + elif method == "kendall": + corr_func = _kendall + else: + corr_func = method + + correlation = np.full(len(variables), np.nan) + for i in range(len(variables)): + rows = ~np.isnan(values[:, i]) + if rows.all(): + correlation[i] = corr_func(values[:, i], target) + elif rows.any(): + correlation[i] = corr_func(values[rows, i], target[rows]) + return correlation + + +def _pearson(a: np.ndarray, b: np.ndarray) -> float: + # constant variables return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(a, b)[0, 1] + + +def _spearman(a: np.ndarray, b: np.ndarray) -> float: + return spearmanr(a, b)[0] + + +def _kendall(a: np.ndarray, b: np.ndarray) -> float: + return kendalltau(a, b)[0] diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) diff --git a/tests/test_selection/test_smart_correlation_selection.py b/tests/test_selection/test_smart_correlation_selection.py index fe56cbc17..233f4495e 100644 --- a/tests/test_selection/test_smart_correlation_selection.py +++ b/tests/test_selection/test_smart_correlation_selection.py @@ -1,20 +1,49 @@ +import re + import numpy as np import pandas as pd import pytest from sklearn.datasets import make_classification -from sklearn.ensemble import RandomForestClassifier, RandomForestRegressor -from sklearn.model_selection import KFold, GroupKFold +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, KFold from feature_engine.selection import SmartCorrelatedSelection -from tests.estimator_checks.init_params_allowed_values_checks import ( - check_error_param_confirm_variables, - check_error_param_missing_values, -) +from tests.backend_helpers import frame_to_dict, make_series + +# at threshold 0.506: var_b is correlated with var_c and var_d, and var_e with var_f. +VAR_CAR = { + "var_a": [1, -1, 0, 0, 0, 0, 0, 0, 0], + "var_b": [0, 0, 1, -1, 2, -2, 0, 0, 1], + "var_c": [0, 0, 10, -10, 0, 0, 0, 0, 9], + "var_d": [0, 0, 0, 0, 1, -1, 0, 0, 1], + "var_e": [0, 0, 0, 0, 0, -1, 2, 3, 4], + "var_f": [0, 0, 0, 0, 0, -1, 20, 30, 30], +} + +WITH_NAN = { + "var_a": [1, -1, 0, 0, 0, 0, 0, 0], + "var_b": [None, 0, 1, -1, 2, -2, 0, 0], + "var_c": [0, 0, 10, -10, 0, 0, 0, 0], + "var_d": [0, 0, 0, 0, 1, -1, 0, 0], + "var_e": [None, 0, 0, 0, 0, -1, 2, 3], + "var_f": [0, 0, 0, 0, 0, -1, 20, 30], +} + +# pearson, spearman and kendall find different groups at threshold 0.9. +CORR_METHODS = { + "var_a": [1.0, 2, 3, 4, 5, 6, 7, 8, 9, 10], + "var_b": [1.2, 2.1, None, 4.3, 4.9, 6.2, 7.1, 7.8, 9.4, 30], + "var_c": [2.0, 1, 3, 4, 6, 5, 8, 7, 10, 9], + "var_d": [10.0, 9, 8, 7, 6, 5, 4, None, 2, 1], + "var_e": [0.0, 1, 0, 1, 0, 1, 0, 1, 0, 1], +} +CORR_METHODS_TARGET = [0, 0, 0, 1, 0, 1, 1, 1, 1, 1] @pytest.fixture(scope="module") -def df_single(): - # create array with 4 correlated features and 2 independent ones +def data_single_group(): + # var_1, var_2 and var_4 are correlated, var_1 and var_2 above 0.8. X, y = make_classification( n_samples=1000, n_features=6, @@ -24,524 +53,571 @@ def df_single(): class_sep=2, random_state=1, ) - - # trasform array into pandas df - colnames = ["var_" + str(i) for i in range(6)] - X = pd.DataFrame(X, columns=colnames) - - return X, y + data = {f"var_{i}": X[:, i].tolist() for i in range(6)} + data["target"] = y.tolist() + return data @pytest.fixture(scope="module") -def df_var_car(): - # create dataframe with known variance and cardinality: +def data_duplicated(): + X, y = make_classification( + n_samples=1000, + n_features=2, + n_informative=2, + n_redundant=0, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + return { + "var_0": X[:, 0].tolist(), + "var_1": X[:, 1].tolist(), + "var_0_duplicated": X[:, 0].tolist(), + "var_1_duplicated": X[:, 1].tolist(), + "target": y.tolist(), + } - # at threshold 0.506 - # a=> no correlated - # b => correlated with c and d - # c => correlated with b - # d => correlated with b - # e => correlated with f - # f => correlated with e +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) - X = pd.DataFrame( - { - "var_a": [1, -1, 0, 0, 0, 0, 0, 0, 0], - "var_b": [0, 0, 1, -1, 2, -2, 0, 0, 1], - "var_c": [0, 0, 10, -10, 0, 0, 0, 0, 9], - "var_d": [0, 0, 0, 0, 1, -1, 0, 0, 1], - "var_e": [0, 0, 0, 0, 0, -1, 2, 3, 4], - "var_f": [0, 0, 0, 0, 0, -1, 20, 30, 30], - } - ) - return X +# init parameters +@pytest.mark.parametrize("threshold", [3, "0.1", -0, 2, 0, 1, -0.1, 1.5, None, [0.5]]) +def test_error_if_threshold_not_float_between_0_and_1(threshold): + msg = f"`threshold` must be a float between 0 and 1. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + SmartCorrelatedSelection(threshold=threshold) -@pytest.fixture(scope="module") -def df_nan(): - X = pd.DataFrame( - { - "var_a": [1, -1, 0, 0, 0, 0, 0, 0], - "var_b": [np.nan, 0, 1, -1, 2, -2, 0, 0], - "var_c": [0, 0, 10, -10, 0, 0, 0, 0], - "var_d": [0, 0, 0, 0, 1, -1, 0, 0], - "var_e": [np.nan, 0, 0, 0, 0, -1, 2, 3], - "var_f": [0, 0, 0, 0, 0, -1, 20, 30], - } +@pytest.mark.parametrize("missing_values", [2, "hola", False, None, ["raise"]]) +def test_error_if_missing_values_not_permitted(missing_values): + msg = ( + "missing_values takes only values 'raise' or 'ignore'. " + f"Got {missing_values} instead." ) - return X - - -_input_params = [ - (None, "pearson", 0.8, "ignore", "missing_values", False), - ("var1", "kendall", 0.5, "raise", "cardinality", True), - (["var1", "var2"], "spearman", 0.4, "raise", "variance", False), -] + with pytest.raises(ValueError, match=re.escape(msg)): + SmartCorrelatedSelection(missing_values=missing_values) @pytest.mark.parametrize( - "_variables, _method, _threshold, _missing_values, _sel_method, _confirm_vars", - _input_params, + "selection_method", [3, "hola", ["cardinality"], None, ("variance",)] ) -def test_input_params_assignment( - _variables, _method, _threshold, _missing_values, _sel_method, _confirm_vars -): - sel = SmartCorrelatedSelection( - variables=_variables, - method=_method, - threshold=_threshold, - missing_values=_missing_values, - selection_method=_sel_method, - confirm_variables=_confirm_vars, - ) - - assert sel.variables == _variables - assert sel.method == _method - assert sel.threshold == _threshold - assert sel.missing_values == _missing_values - assert sel.selection_method == _sel_method - assert sel.confirm_variables == _confirm_vars - - -@pytest.mark.parametrize("_threshold", [3, "0.1", -0, 2, 0, 3, 1]) -def test_raises_error_when_threshold_not_permitted(_threshold): - msg = f"`threshold` must be a float between 0 and 1. Got {_threshold} instead." - with pytest.raises(ValueError) as record: - SmartCorrelatedSelection(threshold=_threshold) - assert record.value.args[0] == msg - - -@pytest.mark.parametrize("_method", [3, "hola", ["cardinality"]]) -def test_raises_error_when_selection_method_not_permitted(_method): +def test_error_if_selection_method_not_permitted(selection_method): msg = ( "selection_method takes only values 'missing_values', 'cardinality', " - f"'variance', 'model_performance' or 'corr_with_target'. Got {_method} instead." + "'variance', 'model_performance' or 'corr_with_target'. " + f"Got {selection_method} instead." ) - with pytest.raises(ValueError) as record: - SmartCorrelatedSelection(selection_method=_method) - assert record.value.args[0] == msg + with pytest.raises(ValueError, match=re.escape(msg)): + SmartCorrelatedSelection(selection_method=selection_method) -def test_raises_error_when_selection_method_performance_and_estimator_none(): +def test_error_if_model_performance_and_estimator_is_none(): msg = ( "Please provide an estimator, e.g., " "RandomForestClassifier or select another " "selection_method." ) - with pytest.raises(ValueError) as record: + with pytest.raises(ValueError, match=re.escape(msg)): SmartCorrelatedSelection(selection_method="model_performance", estimator=None) - assert record.value.args[0] == msg - -def test_error_param_missing_values(): - check_error_param_missing_values(SmartCorrelatedSelection()) +def test_error_if_selection_method_missing_values_and_missing_values_raise(): + msg = ( + "When `selection_method = 'missing_values'`, you need to set " + "`missing_values` to `'ignore'`. Got raise instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + SmartCorrelatedSelection( + missing_values="raise", selection_method="missing_values" + ) -def test_error_param_confirm_variables(): - check_error_param_confirm_variables(SmartCorrelatedSelection()) +@pytest.mark.parametrize("confirm_variables", [2, "hola", [True], None]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + SmartCorrelatedSelection(confirm_variables=confirm_variables) -def test_model_performance_single_corr_group(df_single): - X, y = df_single - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.8, - missing_values="raise", - selection_method="model_performance", - estimator=RandomForestClassifier(n_estimators=10, random_state=1), - scoring="roc_auc", - cv=3, +@pytest.mark.parametrize( + "method, threshold, missing_values, selection_method, estimator, scoring, cv, " + "groups, confirm_variables", + [ + ("pearson", 0.8, "ignore", "missing_values", None, "roc_auc", 3, None, False), + ("kendall", 0.5, "raise", "cardinality", None, "accuracy", 5, [1, 2], True), + ("spearman", 0.4, "raise", "variance", None, "r2", KFold(), None, False), + ( + np.corrcoef, + 0.9, + "ignore", + "model_performance", + LogisticRegression(), + "roc_auc", + GroupKFold(), + [1, 1, 2], + False, + ), + ("pearson", 0.7, "raise", "corr_with_target", None, "roc_auc", 3, None, True), + ], +) +def test_init_param_assignment( + method, + threshold, + missing_values, + selection_method, + estimator, + scoring, + cv, + groups, + confirm_variables, +): + sel = SmartCorrelatedSelection( + method=method, + threshold=threshold, + missing_values=missing_values, + selection_method=selection_method, + estimator=estimator, + scoring=scoring, + cv=cv, + groups=groups, + confirm_variables=confirm_variables, ) + assert sel.method is method + assert sel.threshold == threshold + assert sel.missing_values == missing_values + assert sel.selection_method == selection_method + assert sel.estimator is estimator + assert sel.scoring == scoring + assert sel.cv is cv + assert sel.groups == groups + assert sel.confirm_variables is confirm_variables + + +# fit and transform +def test_selection_method_missing_values(make_df): + # missing values: var_b and var_e 1, all other variables 0. + X = make_df(WITH_NAN) + sel = SmartCorrelatedSelection(threshold=0.4, selection_method="missing_values") + Xt = sel.fit_transform(X) + + assert sel.variables_ == ["var_a", "var_b", "var_c", "var_d", "var_e", "var_f"] + assert sel.correlated_feature_sets_ == [{"var_b", "var_c"}, {"var_e", "var_f"}] + assert sel.correlated_feature_dict_ == {"var_c": {"var_b"}, "var_f": {"var_e"}} + assert sel.features_to_drop_ == ["var_b", "var_e"] + assert sel.feature_names_in_ == list(WITH_NAN) + assert sel.n_features_in_ == 6 + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + k: v for k, v in WITH_NAN.items() if k not in ["var_b", "var_e"] + } - Xt = transformer.fit_transform(X, y) - # expected result - df = X[["var_0", "var_2", "var_3", "var_4", "var_5"]].copy() +def test_selection_method_variance(make_df): + # std: var_f 13.73, var_c 5.83, var_e 1.69, var_b 1.17, var_d 0.60, var_a 0.50. + X = make_df(VAR_CAR) + sel = SmartCorrelatedSelection(threshold=0.506, selection_method="variance") + Xt = sel.fit_transform(X) - # test init params - assert transformer.scoring == "roc_auc" - assert transformer.cv == 3 + assert sel.correlated_feature_sets_ == [{"var_e", "var_f"}, {"var_b", "var_c"}] + assert sel.correlated_feature_dict_ == {"var_f": {"var_e"}, "var_c": {"var_b"}} + assert sel.features_to_drop_ == ["var_e", "var_b"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + k: v for k, v in VAR_CAR.items() if k not in ["var_e", "var_b"] + } - # test fit attrs - assert transformer.correlated_feature_sets_ == [{"var_1", "var_2"}] - assert transformer.features_to_drop_ == ["var_1"] - assert transformer.correlated_feature_dict_ == {"var_2": {"var_1"}} - # test transform output - pd.testing.assert_frame_equal(Xt, df) +def test_selection_method_cardinality(make_df): + # cardinality: var_b 5, var_e 5, var_c 4, var_f 4, var_a 3, var_d 3. + X = make_df(VAR_CAR) + sel = SmartCorrelatedSelection(threshold=0.506, selection_method="cardinality") + Xt = sel.fit_transform(X) -def test_model_performance_2_correlated_groups(df_test): - X, y = df_test + assert sel.correlated_feature_sets_ == [ + {"var_b", "var_c", "var_d"}, + {"var_e", "var_f"}, + ] + assert sel.correlated_feature_dict_ == { + "var_b": {"var_c", "var_d"}, + "var_e": {"var_f"}, + } + assert sel.features_to_drop_ == ["var_c", "var_d", "var_f"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "var_a": VAR_CAR["var_a"], + "var_b": VAR_CAR["var_b"], + "var_e": VAR_CAR["var_e"], + } - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.8, - missing_values="raise", - selection_method="model_performance", - estimator=RandomForestClassifier(n_estimators=10, random_state=1), - scoring="roc_auc", - cv=3, - ) - Xt = transformer.fit_transform(X, y) +def test_cardinality_does_not_count_missing_values(make_df): + # var_x has 4 values and var_y 5: missing values are not a category. + X = make_df({"var_x": [1, 2, 3, 4, None, None], "var_y": [1, 2, 3, 4, 5, 5]}) + sel = SmartCorrelatedSelection(selection_method="cardinality").fit(X) - # expected result - df = X[ - ["var_0", "var_1", "var_2", "var_3", "var_5", "var_7", "var_10", "var_11"] - ].copy() + assert sel.correlated_feature_dict_ == {"var_y": {"var_x"}} + assert sel.features_to_drop_ == ["var_x"] - # test fit attrs - assert transformer.correlated_feature_sets_ == [ - {"var_0", "var_8"}, - {"var_4", "var_6", "var_7", "var_9"}, - ] - assert transformer.features_to_drop_ == [ - "var_8", - "var_4", - "var_6", - "var_9", - ] - assert transformer.correlated_feature_dict_ == { - "var_0": {"var_8"}, - "var_7": {"var_4", "var_6", "var_9"}, - } - # test transform output - pd.testing.assert_frame_equal(Xt, df) +def test_selection_method_corr_with_target(make_df, data_single_group): + X, y = _split_target(make_df, data_single_group) + sel = SmartCorrelatedSelection( + missing_values="raise", selection_method="corr_with_target" + ) + Xt = sel.fit_transform(X, y) -def test_model_performance_single_corr_group_duplicated_features(df_single): - """Test selector consistency in case of very similar columns (e.g. duplicated). + assert sel.correlated_feature_sets_ == [{"var_1", "var_2", "var_4"}] + assert sel.correlated_feature_dict_ == {"var_2": {"var_1", "var_4"}} + assert sel.features_to_drop_ == ["var_4", "var_1"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_0", "var_2", "var_3", "var_5"] - This test checks that in case of columns with the same values for the selection - method (for example same correlation with target), the transformer consistently - drops the feature that is is alphabetically bigger instead of selecting one of the - two columns randomly. - """ - X, y = make_classification( - n_samples=1000, - n_features=2, - n_informative=2, - n_redundant=0, - n_clusters_per_class=1, - weights=[0.50], - class_sep=2, - random_state=1, +@pytest.mark.parametrize( + "method, correlated_dict, features_to_drop", + [ + ("pearson", {"var_a": {"var_c", "var_d"}}, ["var_d", "var_c"]), + ( + "spearman", + {"var_a": {"var_b", "var_c", "var_d"}}, + ["var_d", "var_b", "var_c"], + ), + ("kendall", {"var_d": {"var_a", "var_b"}}, ["var_a", "var_b"]), + ], +) +def test_corr_with_target_uses_correlation_method( + make_df, method, correlated_dict, features_to_drop +): + # absolute correlation with the target: + # pearson: var_a 0.782, var_d 0.763, var_c 0.711, var_b 0.468, var_e 0.408 + # spearman: var_a 0.782, var_d 0.779, var_b 0.730, var_c 0.711, var_e 0.408 + # kendall: var_d 0.671, var_a 0.669, var_b 0.629, var_c 0.609, var_e 0.408 + X = make_df(CORR_METHODS) + y = make_series(make_df, CORR_METHODS_TARGET) + sel = SmartCorrelatedSelection( + method=method, threshold=0.9, selection_method="corr_with_target" ) + Xt = sel.fit_transform(X, y) - # trasform array into pandas df - colnames = ["var_" + str(i) for i in range(2)] - X = pd.DataFrame(X, columns=colnames) + assert sel.correlated_feature_dict_ == correlated_dict + assert sel.features_to_drop_ == features_to_drop + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [f for f in CORR_METHODS if f not in features_to_drop] - # Duplicate columns - for col in X.columns: - X[col + "_duplicated"] = X[col] - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.8, - missing_values="raise", - selection_method="corr_with_target", - estimator=RandomForestClassifier(n_estimators=10, random_state=1), - scoring="roc_auc", - cv=3, - ) +def test_corr_with_target_ties_keep_first_feature(make_df, data_duplicated): + X, y = _split_target(make_df, data_duplicated) + sel = SmartCorrelatedSelection(selection_method="corr_with_target") + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X, y) - - # test init params - assert transformer.scoring == "roc_auc" - assert transformer.cv == 3 - - # test fit attrs - assert transformer.correlated_feature_sets_ == ( - [ - { - "var_1", - "var_1_duplicated", - }, - {"var_0_duplicated", "var_0"}, - ] - ) - assert transformer.features_to_drop_ == [ - "var_1_duplicated", - "var_0_duplicated", + assert sel.correlated_feature_sets_ == [ + {"var_1", "var_1_duplicated"}, + {"var_0", "var_0_duplicated"}, ] - assert transformer.correlated_feature_dict_ == { - "var_1": { - "var_1_duplicated", - }, + assert sel.correlated_feature_dict_ == { + "var_1": {"var_1_duplicated"}, "var_0": {"var_0_duplicated"}, } - - # expected result - df = X[["var_0", "var_1"]].copy() - # test transform output - pd.testing.assert_frame_equal(Xt, df) + assert sel.features_to_drop_ == ["var_1_duplicated", "var_0_duplicated"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_0", "var_1"] -def test_cv_generator(df_single): - X, y = df_single - cv = KFold(3) - - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.8, +def test_model_performance_single_group(make_df, data_single_group): + X, y = _split_target(make_df, data_single_group) + sel = SmartCorrelatedSelection( missing_values="raise", selection_method="model_performance", estimator=RandomForestClassifier(n_estimators=10, random_state=1), scoring="roc_auc", - cv=cv.split(X, y), + cv=3, ) + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X, y) + assert sel.correlated_feature_sets_ == [{"var_1", "var_2"}] + assert sel.correlated_feature_dict_ == {"var_2": {"var_1"}} + assert sel.features_to_drop_ == ["var_1"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_0", "var_2", "var_3", "var_4", "var_5"] - df = X[["var_0", "var_2", "var_3", "var_4", "var_5"]].copy() - pd.testing.assert_frame_equal(Xt, df) - -def test_error_if_select_model_performance_and_y_is_none(df_single): - X, y = df_single - - transformer = SmartCorrelatedSelection( +def test_model_performance_two_groups(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + sel = SmartCorrelatedSelection( + missing_values="raise", selection_method="model_performance", estimator=RandomForestClassifier(n_estimators=10, random_state=1), scoring="roc_auc", + cv=3, ) - msg = ( - "When `selection_method = 'model_performance'` y is needed to fit " - "the transformer." - ) - with pytest.raises(ValueError) as record: - transformer.fit(X) - assert record.value.args[0] == msg + Xt = sel.fit_transform(X, y) + assert sel.correlated_feature_sets_ == [ + {"var_0", "var_8"}, + {"var_4", "var_6", "var_7", "var_9"}, + ] + assert sel.correlated_feature_dict_ == { + "var_0": {"var_8"}, + "var_7": {"var_4", "var_6", "var_9"}, + } + assert sel.features_to_drop_ == ["var_8", "var_4", "var_6", "var_9"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [ + "var_0", + "var_1", + "var_2", + "var_3", + "var_5", + "var_7", + "var_10", + "var_11", + ] -def test_error_if_select_corr_with_target_and_y_is_none(df_single): - X, _ = df_single - transformer = SmartCorrelatedSelection( - selection_method="corr_with_target", - ) - msg = ( - "When `selection_method = 'corr_with_target'` y is needed to fit " - "the transformer." - ) - with pytest.raises(ValueError) as record: - transformer.fit(X) - assert record.value.args[0] == msg - - -def test_selection_method_variance(df_var_car): - X = df_var_car - - # std of each variable: - # var_f 13.727507 - # var_c 5.830952 - # var_e 1.691482 - # var_b 1.166667 - # var_d 0.600925 - # var_a 0.500000 - - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.506, - missing_values="raise", - selection_method="variance", - estimator=None, +def test_model_performance_ties_keep_first_feature_alphabetically( + make_df, data_duplicated +): + # duplicated features train equally good models. + X, y = _split_target(make_df, data_duplicated) + sel = SmartCorrelatedSelection( + selection_method="model_performance", + estimator=RandomForestClassifier(n_estimators=10, random_state=1), + cv=3, ) + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X) - - assert transformer.features_to_drop_ == ["var_e", "var_b"] - assert transformer.correlated_feature_dict_ == { - "var_f": {"var_e"}, - "var_c": {"var_b"}, + assert sel.correlated_feature_dict_ == { + "var_0": {"var_0_duplicated"}, + "var_1": {"var_1_duplicated"}, } - # test transform output - pd.testing.assert_frame_equal(Xt, X.drop(["var_e", "var_b"], axis=1)) - + assert sel.features_to_drop_ == ["var_0_duplicated", "var_1_duplicated"] + assert list(Xt.columns) == ["var_0", "var_1"] -def test_selection_method_cardinality(df_var_car): - X = df_var_car - # cardinality of variables: - # var_b 5 - # var_e 5 - # var_c 4 - # var_f 4 - # var_a 3 - # var_d 3 - - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.506, - missing_values="raise", - selection_method="cardinality", - estimator=None, +def test_model_performance_with_cv_generator(make_df, data_single_group): + X, y = _split_target(make_df, data_single_group) + sel = SmartCorrelatedSelection( + selection_method="model_performance", + estimator=RandomForestClassifier(n_estimators=10, random_state=1), + cv=KFold(3).split(np.zeros(len(y)), y), ) + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X) + assert sel.features_to_drop_ == ["var_1"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["var_0", "var_2", "var_3", "var_4", "var_5"] - assert transformer.features_to_drop_ == ["var_c", "var_d", "var_f"] - assert transformer.correlated_feature_dict_ == { - "var_b": {"var_c", "var_d"}, - "var_e": {"var_f"}, - } - # test transform output - pd.testing.assert_frame_equal(Xt, X.drop(["var_c", "var_d", "var_f"], axis=1)) - - -def test_selection_method_missing_values(df_nan): - X = df_nan - - # expected order of the variables: - # var_a 0 - # var_c 0 - # var_d 0 - # var_f 0 - # var_b 1 - # var_e 1 - - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.4, - missing_values="ignore", - selection_method="missing_values", - estimator=None, - ) - Xt = transformer.fit_transform(X) +def test_model_performance_with_groups(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + groups = np.repeat(np.arange(10), 100) + params = dict( + selection_method="model_performance", + estimator=LogisticRegression(), + scoring="roc_auc", + ) + sel = SmartCorrelatedSelection(cv=GroupKFold(3), groups=groups, **params) + sel.fit(X, y) + splits = GroupKFold(3).split(np.zeros(len(y)), y, groups) + sel_splits = SmartCorrelatedSelection(cv=splits, **params).fit(X, y) - assert transformer.features_to_drop_ == ["var_b", "var_e"] - assert transformer.correlated_feature_dict_ == { - "var_c": {"var_b"}, - "var_f": {"var_e"}, + assert sel.correlated_feature_dict_ == { + "var_0": {"var_8"}, + "var_7": {"var_4", "var_6", "var_9"}, } - # test transform output - pd.testing.assert_frame_equal(Xt, X.drop(["var_b", "var_e"], axis=1)) + assert sel.features_to_drop_ == sel_splits.features_to_drop_ -def test_error_when_selection_method_missing_values_and_missing_values_raise(df_na): - msg = ( - "When `selection_method = 'missing_values'`, you need to set " - "`missing_values` to `'ignore'`. Got raise instead." +@pytest.mark.parametrize("target_type", [list, np.array]) +@pytest.mark.parametrize("selection_method", ["corr_with_target", "model_performance"]) +def test_target_as_list_or_array( + make_df, data_single_group, target_type, selection_method +): + X, _ = _split_target(make_df, data_single_group) + y = target_type(data_single_group["target"]) + sel = SmartCorrelatedSelection( + selection_method=selection_method, + estimator=RandomForestClassifier(n_estimators=10, random_state=1), + cv=3, ) - with pytest.raises(ValueError) as record: - SmartCorrelatedSelection( - missing_values="raise", - selection_method="missing_values", - ) - assert record.value.args[0] == msg - + sel.fit(X, y) -def test_raises_error_when_method_not_permitted(df_var_car): - X = df_var_car - transformer = SmartCorrelatedSelection(method="not_valid") - - with pytest.raises(ValueError) as errmsg: - transformer.fit(X) + expected = { + "corr_with_target": {"var_2": {"var_1", "var_4"}}, + "model_performance": {"var_2": {"var_1"}}, + } + assert sel.correlated_feature_dict_ == expected[selection_method] - exceptionmsg = errmsg.value.args[0] - assert ( - exceptionmsg - == "method must be either 'pearson', 'spearman', 'kendall', or a callable," - + " 'not_valid' was supplied" +@pytest.mark.parametrize("selection_method", ["model_performance", "corr_with_target"]) +def test_error_if_y_is_none(make_df, data_single_group, selection_method): + X, _ = _split_target(make_df, data_single_group) + sel = SmartCorrelatedSelection( + selection_method=selection_method, estimator=LogisticRegression() ) + msg = ( + f"When `selection_method = '{selection_method}'` y is needed to fit " + "the transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + sel.fit(X) -def test_raises_missing_data_error(df_nan): - df = df_nan +@pytest.mark.parametrize("selection_method", ["variance", "corr_with_target"]) +def test_error_if_missing_values_raise_and_nan(make_df, selection_method): + X = make_df(WITH_NAN) + y = make_series(make_df, [0, 1, 0, 1, 0, 1, 0, 1]) + sel = SmartCorrelatedSelection( + selection_method=selection_method, missing_values="raise" + ) msg = ( "Some of the variables in the dataset contain NaN. Check and " "remove those before using this transformer." ) + with pytest.raises(ValueError, match=re.escape(msg)): + sel.fit(X, y) + + +def test_error_if_missing_values_raise_and_inf(make_df): + X = make_df({"var_a": [1.0, float("inf"), 3.0], "var_b": [1.0, 2.0, 3.0]}) sel = SmartCorrelatedSelection(selection_method="variance", missing_values="raise") - with pytest.raises(ValueError) as record: - sel.fit(df) - assert record.value.args[0] == msg + msg = ( + "Some of the variables to transform contain inf values. Check and " + "remove those before using this transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + sel.fit(X) -def test_callable_method(df_test, random_uniform_method): - X, _ = df_test +def test_callable_method(make_df, data_classification): + def abs_pearson(a, b): + return abs(np.corrcoef(a, b)[0, 1]) - transformer = SmartCorrelatedSelection( - method=random_uniform_method, - ) + X, y = _split_target(make_df, data_classification) + sel = SmartCorrelatedSelection(method=abs_pearson, selection_method="variance") + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X) + assert sel.correlated_feature_dict_ == { + "var_7": {"var_4", "var_6", "var_9"}, + "var_8": {"var_0"}, + } + assert sel.features_to_drop_ == ["var_6", "var_4", "var_9", "var_0"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [ + "var_1", + "var_2", + "var_3", + "var_5", + "var_7", + "var_8", + "var_10", + "var_11", + ] - # test no empty dataframe - assert not Xt.empty - # test fit attrs - assert len(transformer.correlated_feature_sets_) > 0 - assert len(transformer.features_to_drop_) > 0 - assert len(transformer.variables_) > 0 - assert transformer.n_features_in_ == len(X.columns) +@pytest.mark.parametrize("confirm_variables", [False, True]) +def test_variables_ignore_other_columns( + make_df, data_classification, confirm_variables +): + data = {k: v for k, v in data_classification.items() if k != "target"} + data["cat"] = ["a", "b"] * 500 + variables = ["var_4", "var_6", "var_7", "var_0"] + if confirm_variables is True: + variables = variables + ["not_in_df"] + X = make_df(data) + sel = SmartCorrelatedSelection( + variables=variables, + selection_method="variance", + confirm_variables=confirm_variables, + ) + Xt = sel.fit_transform(X) + assert sel.variables_ == ["var_4", "var_6", "var_7", "var_0"] + assert sel.features_to_drop_ == ["var_6", "var_4"] + assert isinstance(Xt, make_df) + assert list(Xt.columns) == [c for c in data if c not in ["var_6", "var_4"]] -def test_smart_correlation_selection_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups - cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) - estimator = RandomForestRegressor(n_estimators=3, random_state=1) - scoring = "neg_mean_absolute_error" - selection_method = "variance" +def test_fit_does_not_modify_input(make_df, data_single_group): + X, y = _split_target(make_df, data_single_group) + SmartCorrelatedSelection(selection_method="corr_with_target").fit(X, y) + X_original, _ = _split_target(make_df, data_single_group) + assert frame_to_dict(X) == frame_to_dict(X_original) - transformer_expected = SmartCorrelatedSelection( - estimator=estimator, - scoring=scoring, - selection_method=selection_method, - cv=cv_indices, + +def test_error_if_method_not_permitted(): + # the error comes from pandas. + X = pd.DataFrame(VAR_CAR) + sel = SmartCorrelatedSelection(method="not_valid") + msg = ( + "method must be either 'pearson', 'spearman', 'kendall', or a callable, " + "'not_valid' was supplied" ) + with pytest.raises(ValueError, match=re.escape(msg)): + sel.fit(X) - X_tr_expected = transformer_expected.fit_transform(X, y) - transformer = SmartCorrelatedSelection( - estimator=estimator, - scoring=scoring, - selection_method=selection_method, - cv=cv, - groups=groups, +def test_error_if_index_of_X_and_y_differ(data_single_group): + X = pd.DataFrame( + {k: v for k, v in data_single_group.items() if k != "target"}, + index=range(10, 1010), ) + y = pd.Series(data_single_group["target"]) + sel = SmartCorrelatedSelection(selection_method="corr_with_target") + with pytest.raises( + ValueError, match=re.escape("The indexes of X and y do not match.") + ): + sel.fit(X, y) - X_tr = transformer.fit_transform(X, y) - - pd.testing.assert_frame_equal(X_tr_expected, X_tr) +@pytest.mark.parametrize( + "selection_method", + ["missing_values", "cardinality", "variance", "corr_with_target"], +) +def test_integer_column_names(selection_method): + X = pd.DataFrame({i: VAR_CAR[f"var_{c}"] for i, c in enumerate("abcdef")}) + y = pd.Series([0, 1, 0, 1, 0, 1, 1, 1, 1]) + sel = SmartCorrelatedSelection(threshold=0.506, selection_method=selection_method) + Xt = sel.fit_transform(X, y) + + expected = { + "missing_values": [2, 3, 5], + "cardinality": [2, 3, 5], + "variance": [4, 1], + "corr_with_target": [2, 3, 4], + } + assert sel.features_to_drop_ == expected[selection_method] + pd.testing.assert_frame_equal(Xt, X.drop(columns=expected[selection_method])) -def test_corr_with_target_single_corr_group(df_single): - X, y = df_single - transformer = SmartCorrelatedSelection( - variables=None, - method="pearson", - threshold=0.8, - missing_values="raise", - selection_method="corr_with_target", +def test_model_performance_with_integer_column_names(data_single_group): + X = pd.DataFrame({i: data_single_group[f"var_{i}"] for i in range(6)}) + y = pd.Series(data_single_group["target"]) + sel = SmartCorrelatedSelection( + selection_method="model_performance", + estimator=RandomForestClassifier(n_estimators=10, random_state=1), + cv=3, ) + Xt = sel.fit_transform(X, y) - Xt = transformer.fit_transform(X, y) + assert sel.correlated_feature_dict_ == {2: {1}} + pd.testing.assert_frame_equal(Xt, X[[0, 2, 3, 4, 5]]) - # expected result - df = X[["var_0", "var_2", "var_3", "var_5"]].copy() - # test fit attrs - assert transformer.correlated_feature_sets_ == [{"var_1", "var_2", "var_4"}] - assert transformer.features_to_drop_ == ["var_4", "var_1"] - assert transformer.correlated_feature_dict_ == {"var_2": {"var_1", "var_4"}} - # test transform output - pd.testing.assert_frame_equal(Xt, df) +def test_transform_keeps_pandas_index(data_single_group): + X = pd.DataFrame( + {k: v for k, v in data_single_group.items() if k != "target"}, + index=range(10, 1010), + ) + y = pd.Series(data_single_group["target"], index=X.index) + Xt = SmartCorrelatedSelection(selection_method="corr_with_target").fit_transform( + X, y + ) + pd.testing.assert_frame_equal(Xt, X[["var_0", "var_2", "var_3", "var_5"]])