diff --git a/docs/user_guide/selection/DropConstantFeatures.rst b/docs/user_guide/selection/DropConstantFeatures.rst index e6e90c0c2..d95a13c96 100644 --- a/docs/user_guide/selection/DropConstantFeatures.rst +++ b/docs/user_guide/selection/DropConstantFeatures.rst @@ -71,8 +71,8 @@ Next, we load the Titanic dataset and separate it into a training set and a test ) Now, we set up the :class:`DropConstantFeatures()` to remove features that show the same -value in more than 70% of the observations. We do this through the parameter `tol`. The -default value for this parameter is zero, in which case it will remove constant features. +value in 70% or more of the observations. We do this through the parameter `tol`. The +default value for this parameter is 1, in which case it will remove constant features. .. code:: python @@ -93,7 +93,7 @@ The variables to drop are stored in the attribute `features_to_drop_`: transformer.features_to_drop_ -These are the 4 features that show the same value in more than 70% of the rows: +These are the 4 features that show the same value in 70% or more of the rows: .. code:: python @@ -115,7 +115,7 @@ We obtain the following proportions: C 0.195415 Q 0.090611 Missing 0.002183 - Name: embarked, dtype: float64 + Name: proportion, dtype: float64 Based on the previous results, 71% of the passengers embarked in S. @@ -138,7 +138,7 @@ We obtain the following proportions: 5 0.003275 6 0.002183 9 0.001092 - Name: parch, dtype: float64 + Name: proportion, dtype: float64 Based on the previous results, 77% of the passengers had 0 parent or child. Because of this, these features were deemed quasi-constant and will be removed in the next step. @@ -215,6 +215,68 @@ This and other feature selection methods may not necessarily avoid overfitting, contribute to simplifying our machine learning pipelines and creating more interpretable machine learning models. +Missing values +-------------- + +By default, :class:`DropConstantFeatures()` raises an error if the variables contain +missing values. With `missing_values="include"`, missing values are counted as one more +value of the variable. With `missing_values="ignore"`, missing values are not counted as a +value, but the proportion of the most frequent value is still calculated over all the rows. +In addition, with `tol=1` and `missing_values="ignore"`, variables that show a single value +besides the missing data are dropped. + +With polars +----------- + +:class:`DropConstantFeatures()` also works with polars dataframes, and returns a polars +dataframe. Both null and NaN are treated as missing values: + +.. code:: python + + import polars as pl + from feature_engine.selection import DropConstantFeatures + + X = pl.DataFrame({ + "city": ["London", "London", "London", "London", "Paris"], + "rooms": [3, 2, None, 4, 3], + "garden": [True, True, True, True, True], + "floor": [1.0, 1.0, None, None, 1.0], + }) + + dcf = DropConstantFeatures(tol=0.8, missing_values="ignore") + Xt = dcf.fit_transform(X) + + print(dcf.features_to_drop_) + +The variables `city` and `garden` show the same value in 80% or more of the rows. The +variable `floor` shows the value 1.0 in 3 of 5 rows, 60%, because the missing values count +towards the total number of rows: + +.. code:: python + + ['city', 'garden'] + +The transformed dataframe is a polars dataframe: + +.. code:: python + + print(Xt) + +.. code:: text + + shape: (5, 2) + ┌───────┬───────┐ + │ rooms ┆ floor │ + │ --- ┆ --- │ + │ i64 ┆ f64 │ + ╞═══════╪═══════╡ + │ 3 ┆ 1.0 │ + │ 2 ┆ 1.0 │ + │ null ┆ null │ + │ 4 ┆ null │ + │ 3 ┆ 1.0 │ + └───────┴───────┘ + Additional resources -------------------- diff --git a/feature_engine/selection/base_recursive_selector.py b/feature_engine/selection/base_recursive_selector.py index fe9113077..1161572aa 100644 --- a/feature_engine/selection/base_recursive_selector.py +++ b/feature_engine/selection/base_recursive_selector.py @@ -1,7 +1,9 @@ from types import GeneratorType -from typing import List, Union +from typing import List, Tuple, Union -import pandas as pd +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame, IntoSeries from sklearn.inspection import permutation_importance from sklearn.model_selection import cross_validate @@ -9,14 +11,13 @@ _check_variables_input_value, ) from feature_engine.dataframe_checks import check_X_y -from feature_engine.selection.base_selection_functions import get_feature_importances +from feature_engine.selection.base_selection_functions import ( + _importance_series, + _select_numerical_variables, + get_feature_importances, +) from feature_engine.selection.base_selector import BaseSelector from feature_engine.tags import _return_tags -from feature_engine.variable_handling import ( - check_numerical_variables, - find_numerical_variables, - retain_variables_if_in_df, -) Variables = Union[None, int, str, List[Union[str, int]]] @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector): Performance of the model trained using the original dataset. feature_importances_: - Pandas Series with the feature importance (comes from step 2) + The feature importance (comes from step 2). A pandas Series with the + features as index when X is a pandas dataframe, and a dictionary with the + features as keys otherwise. feature_importances_std_: - Pandas Series with the standard deviation of the feature importance. + The standard deviation of the feature importance, as a pandas Series or a + dictionary, like `feature_importances_`. features_to_drop_: List with the features to remove from the dataset. @@ -116,7 +120,9 @@ def __init__( ): if not isinstance(threshold, (int, float)): - raise ValueError("threshold can only be integer or float") + raise ValueError( + f"threshold must be an integer or a float. Got {threshold} instead." + ) super().__init__(confirm_variables) self.variables = _check_variables_input_value(variables) @@ -126,43 +132,45 @@ def __init__( self.cv = cv self.groups = groups - def fit(self, X: pd.DataFrame, y: pd.Series): + def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]: """ Find initial model performance. Sort features by importance. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) Target variable. Required to train the estimator. - """ - # check input dataframe - X, y = check_X_y(X, y) + Returns + ------- + nw_X: narwhals dataframe + The input dataframe, as a narwhals dataframe. - if self.variables is None: - self.variables_ = find_numerical_variables(X) - else: - if self.confirm_variables is True: - variables_ = retain_variables_if_in_df(X, self.variables) - self.variables_ = check_numerical_variables(X, variables_) - else: - self.variables_ = check_numerical_variables(X, self.variables) + y: Series or numpy array + The target, checked. + """ + nw_X, y = check_X_y(X, y) + + self.variables_ = _select_numerical_variables( + X, self.variables, self.confirm_variables + ) self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv # check that there are more than 1 variable to select from self._check_variable_number() - # save input features self._get_feature_names_in(X) + X_model = nw_X.select(nw.col(*self.variables_)).to_native() + # train model with all features and cross-validation model = cross_validate( estimator=self.estimator, - X=X[self.variables_], + X=X_model, y=y, cv=self._cv, groups=self.groups, @@ -170,40 +178,32 @@ def fit(self, X: pd.DataFrame, y: pd.Series): return_estimator=True, ) - # store initial model performance self.initial_model_performance_ = model["test_score"].mean() - # Initialize a dataframe that will contain the list of the feature/coeff - # importance for each cross validation fold - feature_importances_cv = pd.DataFrame() - - # Populate the feature_importances_cv dataframe with columns containing - # the feature importance values for each model returned by the cross - # validation. - # There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - + # one row of feature importance per cross-validation fold + importances = [] + for m in model["estimator"]: if hasattr(m, "feature_importances_") or hasattr(m, "coef_"): - feature_importances_cv[i] = get_feature_importances(m) + importances.append(get_feature_importances(m)) else: r = permutation_importance( m, - X[self.variables_], + X_model, y, n_repeats=1, random_state=10, ) - feature_importances_cv[i] = r.importances_mean + importances.append(r.importances_mean) + importances_arr = np.array(importances) - # Add the variables as index to feature_importances_cv - feature_importances_cv.index = self.variables_ - - # Aggregate the feature importance returned in each fold - self.feature_importances_ = feature_importances_cv.mean(axis=1) - self.feature_importances_std_ = feature_importances_cv.std(axis=1) + self.feature_importances_ = _importance_series( + X, self.variables_, importances_arr.mean(axis=0) + ) + self.feature_importances_std_ = _importance_series( + X, self.variables_, importances_arr.std(axis=0, ddof=1) + ) - return X, y + return nw_X, y def _more_tags(self): tags_dict = _return_tags() diff --git a/feature_engine/selection/base_selection_functions.py b/feature_engine/selection/base_selection_functions.py index 96fc45970..4cbadb60b 100644 --- a/feature_engine/selection/base_selection_functions.py +++ b/feature_engine/selection/base_selection_functions.py @@ -1,8 +1,11 @@ -from typing import List, Union from types import GeneratorType +from typing import List, Union +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame +from scipy.stats import kendalltau, rankdata from sklearn.model_selection import cross_validate from feature_engine.variable_handling import ( @@ -36,8 +39,19 @@ def get_feature_importances(estimator): return importances +def _importance_series(X: IntoDataFrame, features, values: np.ndarray): + """ + Return the importance of each feature as a pandas Series indexed by the + features when X is a pandas dataframe, or as a dictionary with the features as + keys otherwise. + """ + if nwd.is_pandas_dataframe(X) is True: + return nw.get_native_namespace(X).Series(values, index=features) + return dict(zip(features, values.tolist())) + + def _select_all_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, exclude_datetime: bool = False, @@ -62,7 +76,7 @@ def _select_all_variables( def _select_numerical_variables( - X: pd.DataFrame, + X: IntoDataFrame, variables: Variables, confirm_variables: bool, ): @@ -85,8 +99,113 @@ def _select_numerical_variables( return variables_ +def _corrcoef(values: np.ndarray) -> np.ndarray: + # constant columns return NaN, like pandas, instead of warning. + with np.errstate(divide="ignore", invalid="ignore"): + return np.corrcoef(values, rowvar=False) + + +def _pearson_pairwise_complete(values: np.ndarray, finite: np.ndarray) -> np.ndarray: + """ + Pearson correlation of every pair of columns, using the rows where both are + finite. The sums of all pairs come from matrix products, which is much faster + than looping over the pairs. + """ + mask: np.ndarray = finite.astype(float) + with np.errstate(divide="ignore", invalid="ignore"): + # centring first keeps the sums below numerically stable. + mean = np.where(finite, values, 0.0).sum(axis=0) / mask.sum(axis=0) + x = np.where(finite, values - mean, 0.0) + n_obs = mask.T @ mask + sum_x = x.T @ mask + sum_xx = (x * x).T @ mask + var = sum_xx - sum_x * sum_x / n_obs + corr = (x.T @ x - sum_x * sum_x.T / n_obs) / np.sqrt(var * var.T) + + # when the variance of the shared rows is tiny compared with the sums, the + # subtraction above loses precision: recompute those pairs one by one. + unstable = (var <= 1e-8 * sum_xx) | (var.T <= 1e-8 * sum_xx.T) + corr[unstable] = np.nan + for i, j in zip(*np.nonzero(np.triu(unstable & (n_obs > 1), 1))): + rows = finite[:, i] & finite[:, j] + corr[i, j] = _corrcoef(x[rows][:, [i, j]])[0, 1] + return corr + + +def _spearman_pairwise_complete(nw_X: nw.DataFrame) -> np.ndarray: + """ + Spearman correlation of every pair of columns, ranking each pair on the rows + where both are finite, like pandas.DataFrame.corr(). + """ + variables = nw_X.columns + exprs = [] + for i, var_i in enumerate(variables): + for j in range(i + 1, len(variables)): + var_j = variables[j] + rows = nw.col(var_i).is_finite() & nw.col(var_j).is_finite() + x = nw.when(rows).then(nw.col(var_i)).rank("average") + y = nw.when(rows).then(nw.col(var_j)).rank("average") + dx = x - x.mean() + dy = y - y.mean() + exprs.append( + ((dx * dy).sum() / ((dx * dx).sum() * (dy * dy).sum()).sqrt()).alias( + f"__{i}_{j}__" + ) + ) + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + corr[np.triu_indices(n_vars, 1)] = np.array(nw_X.select(exprs).row(0), dtype=float) + return corr + + +def _correlation_matrix(X: IntoDataFrame, variables: list, method) -> np.ndarray: + """ + Correlation matrix of the variables. Like pandas.DataFrame.corr(), each pair of + variables is compared on the rows where both have finite values. Only the + values above the diagonal are used. + """ + if nwd.is_pandas_dataframe(X) is True: + values = X[variables].to_numpy(dtype=float, na_value=np.nan) + else: + nw_X = nw.from_native(X, eager_only=True).select(nw.col(*variables)) + values = nw_X.to_numpy().astype(float) + finite = np.isfinite(values) + + # numpy is faster than pandas and narwhals. + if method == "pearson": + if finite.all(): + return _corrcoef(values) + return _pearson_pairwise_complete(values, finite) + + if method == "spearman" and finite.all(): + # scipy ranks faster than pandas, and polars faster than scipy. + if nwd.is_pandas_dataframe(X) is True: + return _corrcoef(rankdata(values, axis=0)) + return _corrcoef(nw_X.select(nw.all().rank("average")).to_numpy()) + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[variables].corr(method=method).to_numpy() + + if method == "spearman": + return _spearman_pairwise_complete(nw_X) + + # kendall and callables are computed pair by pair, as pandas does. + corr_func = (lambda a, b: kendalltau(a, b)[0]) if method == "kendall" else method + n_vars = len(variables) + corr = np.full((n_vars, n_vars), np.nan) + for i in range(n_vars): + for j in range(i + 1, n_vars): + rows = finite[:, i] & finite[:, j] + if rows.all(): + corr[i, j] = corr_func(values[:, i], values[:, j]) + elif rows.any(): + corr[i, j] = corr_func(values[rows, i], values[rows, j]) + return corr + + def find_correlated_features( - X: pd.DataFrame, + X: IntoDataFrame, variables: list[Union[str, int]], method: str, threshold: float, @@ -96,7 +215,7 @@ def find_correlated_features( Parameters ---------- - X : pandas dataframe of shape = [n_samples, n_features] + X : dataframe of shape = [n_samples, n_features] The training dataset. variables : list @@ -133,36 +252,32 @@ def find_correlated_features( correlated with the key. The key + the values should be the same as the set found in `correlated_feature_groups`. """ - # the correlation matrix - correlated_matrix = X[variables].corr(method=method).to_numpy() + correlated_matrix = _correlation_matrix(X, variables, method) # the correlated pairs correlated_mask = np.triu(np.abs(correlated_matrix), 1) > threshold - examined = set() + examined: np.ndarray = np.zeros(len(variables), dtype=bool) correlated_groups = list() features_to_drop = list() correlated_dict = {} for i, f_i in enumerate(variables): - if f_i not in examined: - examined.add(f_i) - temp_set = set([f_i]) - for j, f_j in enumerate(variables): - if f_j not in examined: - if correlated_mask[i, j] == 1: - examined.add(f_j) - features_to_drop.append(f_j) - temp_set.add(f_j) - if len(temp_set) > 1: - correlated_groups.append(temp_set) - correlated_dict[f_i] = temp_set.difference({f_i}) + if examined.item(i) is False: + examined[i] = True + correlated = np.flatnonzero(correlated_mask[i] & ~examined) + if len(correlated) > 0: + examined[correlated] = True + correlated_features = [variables[j] for j in correlated] + features_to_drop.extend(correlated_features) + correlated_groups.append({f_i, *correlated_features}) + correlated_dict[f_i] = set(correlated_features) return correlated_groups, features_to_drop, correlated_dict def single_feature_performance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, variables: List[Union[str, int]], estimator, cv, @@ -174,7 +289,7 @@ def single_feature_performance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -211,12 +326,13 @@ def single_feature_performance( feature_performance_std = {} cv = list(cv) if isinstance(cv, GeneratorType) else cv + nw_X = nw.from_native(X, eager_only=True) # train a model for every feature and store the performance for feature in variables: model = cross_validate( estimator, - X[feature].to_frame(), + nw_X.get_column(feature).to_frame().to_native(), y, cv=cv, groups=groups, @@ -230,8 +346,8 @@ def single_feature_performance( def find_feature_importance( - X: pd.DataFrame, - y: pd.Series, + X: IntoDataFrame, + y, estimator, cv, scoring, @@ -245,7 +361,7 @@ def find_feature_importance( Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The input dataframe y: array-like of shape (n_samples) @@ -268,14 +384,15 @@ def find_feature_importance( Returns ------- - feature_importance: pd.Series - A pandas Series with the feature name as index and its importance as value. The - importance is given by the coefficients of linear models or the impurity gain - from tree-based models. - - feature_importance_std: pd.Series - A pandas Series with the feature name as key and the standard deviation of the - feature importance as value. + feature_importance: pandas Series or dict + The importance of each feature, given by the coefficients of linear models or + the impurity gain from tree-based models. A pandas Series with the feature + names as index when X is a pandas dataframe, and a dictionary with the + feature names as keys otherwise. + + feature_importance_std: pandas Series or dict + The standard deviation of the importance of each feature, as a pandas Series + or a dictionary, like `feature_importance`. """ cv = list(cv) if isinstance(cv, GeneratorType) else cv @@ -289,19 +406,16 @@ def find_feature_importance( return_estimator=True, ) - # dataframe to store the feature importance for each cv fold - feature_importances_cv = pd.DataFrame() - - # Populate dataframe with columns containing the feature importance values - # for each cv fold. There are as many columns as folds. - for i in range(len(model["estimator"])): - m = model["estimator"][i] - feature_importances_cv[i] = get_feature_importances(m) + importances = np.array([get_feature_importances(m) for m in model["estimator"]]) - # add the variables as the index to feature_importances_cv - feature_importances_cv.index = X.columns + # pandas keeps the columns index, with its name and dtype. + if nwd.is_pandas_dataframe(X) is True: + features = X.columns + else: + features = nw.from_native(X, eager_only=True).columns - # aggregate the feature importance returned in each fold - feature_importances_ = feature_importances_cv.mean(axis=1) - feature_importances_std_ = feature_importances_cv.std(axis=1) + feature_importances_ = _importance_series(X, features, importances.mean(axis=0)) + feature_importances_std_ = _importance_series( + X, features, importances.std(axis=0, ddof=1) + ) return feature_importances_, feature_importances_std_ diff --git a/feature_engine/selection/base_selector.py b/feature_engine/selection/base_selector.py index cfa8f1c95..0e4ff5d97 100644 --- a/feature_engine/selection/base_selector.py +++ b/feature_engine/selection/base_selector.py @@ -1,5 +1,7 @@ +import narwhals as nw +import narwhals.dependencies as nwd import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -41,41 +43,43 @@ def __init__( self.confirm_variables = confirm_variables - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Return dataframe with selected features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features]. + X: dataframe of shape = [n_samples, n_features]. The input dataframe. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_selected_features] - Pandas dataframe with the selected features. + X_new: dataframe of shape = [n_samples, n_selected_features] + The dataframe with the selected features, in the same library as the + input. """ - - # check if fit is performed prior to transform check_is_fitted(self) - - # check if input is a dataframe - X = check_X(X) - - # check if number of columns in test dataset matches to train dataset + nw_X = check_X(X) _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - X = X[self.feature_names_in_] + # selecting in the train set order also restores the train column order. + features_to_drop = set(self.features_to_drop_) + features = [f for f in self.feature_names_in_ if f not in features_to_drop] - # return the dataframe with the selected features - return X.drop(columns=self.features_to_drop_) + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + return X[features] + else: + return nw_X.select(nw.col(*features)).to_native() - def _get_feature_names_in(self, X): - """Get the names and number of features in the train set. The dataframe - used during fit.""" + def _get_feature_names_in(self, X: IntoDataFrame): + """Get the names and number of features in the train set (the dataframe + used during fit).""" - self.feature_names_in_ = X.columns.to_list() + if nwd.is_pandas_dataframe(X) is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self diff --git a/feature_engine/selection/drop_constant_features.py b/feature_engine/selection/drop_constant_features.py index 204aabbe7..ea10c84c3 100644 --- a/feature_engine/selection/drop_constant_features.py +++ b/feature_engine/selection/drop_constant_features.py @@ -1,6 +1,8 @@ -from typing import List, Union +from typing import List, Optional, Union -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_variables import ( _check_variables_input_value, @@ -59,15 +61,19 @@ class DropConstantFeatures(BaseSelector): tol: float,int, default=1 Threshold to detect constant/quasi-constant features. Variables showing the - same value in a percentage of observations greater than tol will be considered - constant / quasi-constant and dropped. If tol=1, the transformer removes - constant variables. Else, it will remove quasi-constant variables. For example, - if tol=0.98, the transformer will remove variables that show the same value in - 98% of the observations. + same value in a proportion of observations equal to or greater than tol will + be considered constant / quasi-constant and dropped. If tol=1, the + transformer removes constant variables. Else, it will remove quasi-constant + variables. For example, if tol=0.98, the transformer will remove variables + that show the same value in at least 98% of the observations. missing_values: str, default='raise' Whether the missing values should be raised as error, ignored or included as an additional value of the variable. Takes values 'raise', 'ignore', 'include'. + With 'ignore', the proportion of the most frequent value is still calculated + over all observations, and, if tol=1, variables with a single value besides + the missing values are dropped. NaN and null are both treated as missing + values. {confirm_variables} @@ -113,7 +119,7 @@ class DropConstantFeatures(BaseSelector): >>> x3 = [True, False, False, True])) >>> dcf = DropConstantFeatures() >>> dcf.fit_transform(X) - x2 x3 + x2 x3 0 a True 1 a False 2 b False @@ -126,11 +132,32 @@ class DropConstantFeatures(BaseSelector): >>> x3 = [True, False, False, False])) >>> dcf = DropConstantFeatures(tol = 0.75) >>> dcf.fit_transform(X) - x2 + x2 0 a 1 a 2 b 3 c + + With polars: + + >>> import polars as pl + >>> from feature_engine.selection import DropConstantFeatures + >>> X = pl.DataFrame(dict(x1 = [1,1,1,1], + >>> x2 = ["a", "a", "b", "c"], + >>> x3 = [True, False, False, False])) + >>> dcf = DropConstantFeatures(tol = 0.75) + >>> dcf.fit_transform(X) + shape: (4, 1) + ┌─────┐ + │ x2 │ + │ --- │ + │ str │ + ╞═════╡ + │ a │ + │ a │ + │ b │ + │ c │ + └─────┘ """ def __init__( @@ -147,11 +174,18 @@ def __init__( or tol < 0 or tol > 1 ): - raise ValueError("tol must be a float or integer between 0 and 1") + raise ValueError( + f"tol must be a float or integer between 0 and 1. Got {tol} instead." + ) - if missing_values not in ["raise", "ignore", "include"]: + if not isinstance(missing_values, str) or missing_values not in [ + "raise", + "ignore", + "include", + ]: raise ValueError( - "missing_values takes only values 'raise', 'ignore' or " "'include'." + "missing_values takes only values 'raise', 'ignore' or 'include'. " + f"Got {missing_values} instead." ) super().__init__(confirm_variables) @@ -160,20 +194,21 @@ def __init__( self.variables = _check_variables_input_value(variables) self.missing_values = missing_values - def fit(self, X: pd.DataFrame, y: pd.Series = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Find constant and quasi-constant features. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] - The input dataframe. + X: dataframe of shape = [n_samples, n_features] + The input dataframe. Can be a pandas, polars, or any other dataframe + supported by narwhals. y: None y is not needed for this transformer. You can pass y or None. """ # check input dataframe - X = check_X(X) + nw_X = check_X(X) self.variables_ = _select_all_variables( X, self.variables, self.confirm_variables @@ -183,32 +218,66 @@ def fit(self, X: pd.DataFrame, y: pd.Series = None): # check if dataset contains na _check_contains_na(X, self.variables_) - if self.missing_values == "include": - X[self.variables_] = X[self.variables_].fillna("missing_values") + dropna = self.missing_values != "include" + n_rows = nw_X.shape[0] + + # pandas is faster than narwhals. + if nwd.is_pandas_dataframe(X) is True: + if self.tol == 1: + self.features_to_drop_ = [ + feature + for feature in self.variables_ + if X[feature].nunique(dropna=dropna) == 1 + ] + else: + # variables with only missing values have no counts; max() is NaN + # and they are kept. + self.features_to_drop_ = [ + feature + for feature in self.variables_ + if X[feature].value_counts(dropna=dropna, sort=False).max() / n_rows + >= self.tol + ] - # find constant features - if self.tol == 1: - self.features_to_drop_ = [ - feature for feature in self.variables_ if X[feature].nunique() == 1 - ] - - # find constant and quasi-constant features else: - self.features_to_drop_ = [] - - for feature in self.variables_: - # find most frequent value / category in the variable - predominant = ( - (X[feature].value_counts() / float(len(X))) - .sort_values(ascending=False) - .values[0] - ) + float_vars = [f for f in self.variables_ if nw_X.schema[f].is_float()] + + if self.tol == 1: + n_unique = nw_X.select( + _missing_values(nw.col(f), f in float_vars, dropna).n_unique() + for f in self.variables_ + ).row(0) + drop = [n == 1 for n in n_unique] + + else: + if nwd.is_polars_dataframe(X) is True: + # polars counts the values of all variables in parallel, which is + # faster than narwhals. + native_ns = nw.get_native_namespace(nw_X) + counts = X.select( + _missing_values(native_ns.col(f), f in float_vars, dropna) + .unique_counts() + .max() + for f in self.variables_ + ).row(0) + else: + counts = [ + _missing_values(nw_X.get_column(f), f in float_vars, dropna) + .value_counts(sort=False, name="__count__") + .get_column("__count__") + .max() + for f in self.variables_ + ] + # variables with only missing values have no counts (None) and are + # kept. + drop = [c is not None and c / n_rows >= self.tol for c in counts] - if predominant >= self.tol: - self.features_to_drop_.append(feature) + self.features_to_drop_ = [ + f for f, d in zip(self.variables_, drop) if d is True + ] # check we are not dropping all the columns in the df - if len(self.features_to_drop_) == len(X.columns): + if len(self.features_to_drop_) == nw_X.shape[1]: raise ValueError( "The resulting dataframe will have no columns after dropping all " "constant or quasi-constant features. Try changing the tol value." @@ -233,3 +302,14 @@ def __sklearn_tags__(self): tags = super().__sklearn_tags__() tags.input_tags.allow_nan = True return tags + + +def _missing_values(column, is_float: bool, dropna: bool): + """Treat NaN as missing, like pandas does, and drop the missing values if + dropna is True. Works with narwhals and polars expressions and series.""" + # polars keeps NaN apart from null. + if is_float is True: + column = column.fill_nan(None) + if dropna is True: + column = column.drop_nulls() + return column diff --git a/tests/test_selection/conftest.py b/tests/test_selection/conftest.py index e41d7ce4e..7006979c8 100644 --- a/tests/test_selection/conftest.py +++ b/tests/test_selection/conftest.py @@ -25,6 +25,23 @@ def df_test(): return X, y +@pytest.fixture(scope="module") +def data_classification(): + """The data of df_test as a plain dict, with the target under "target".""" + X, y = make_classification( + n_samples=1000, + n_features=12, + n_redundant=4, + n_clusters_per_class=1, + weights=[0.50], + class_sep=2, + random_state=1, + ) + data = {f"var_{i}": X[:, i].tolist() for i in range(12)} + data["target"] = y.tolist() + return data + + @pytest.fixture(scope="module") def df_test_with_groups(): # Parameters diff --git a/tests/test_selection/test_base_recursive_selector.py b/tests/test_selection/test_base_recursive_selector.py new file mode 100644 index 000000000..a4b0a3c6b --- /dev/null +++ b/tests/test_selection/test_base_recursive_selector.py @@ -0,0 +1,257 @@ +import re + +import narwhals as nw +import numpy as np +import pandas as pd +import pytest +from sklearn.ensemble import RandomForestClassifier +from sklearn.linear_model import LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold +from sklearn.neighbors import KNeighborsClassifier + +from feature_engine.selection.base_recursive_selector import BaseRecursiveSelector +from tests.backend_helpers import make_series + +VARIABLES = ["var_0", "var_4", "var_7"] + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +# init parameters +@pytest.mark.parametrize("threshold", [None, [0.1], "a_string", {"a": 1}]) +def test_error_if_threshold_not_number(threshold): + msg = f"threshold must be an integer or a float. Got {threshold} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + BaseRecursiveSelector(RandomForestClassifier(), threshold=threshold) + + +@pytest.mark.parametrize( + "estimator, scoring, cv, groups, threshold, confirm_variables", + [ + (RandomForestClassifier(), "roc_auc", 3, None, 0.01, False), + (LogisticRegression(), "accuracy", StratifiedKFold(), [1, 2], 1, True), + (KNeighborsClassifier(), "r2", GroupKFold(), None, -0.5, False), + ], +) +def test_init_param_assignment( + estimator, scoring, cv, groups, threshold, confirm_variables +): + selector = BaseRecursiveSelector( + estimator, + scoring=scoring, + cv=cv, + groups=groups, + threshold=threshold, + confirm_variables=confirm_variables, + ) + assert selector.estimator is estimator + assert selector.scoring == scoring + assert selector.cv is cv + assert selector.groups == groups + assert selector.threshold == threshold + assert selector.confirm_variables is confirm_variables + + +# fit +@pytest.mark.parametrize("cv_type", ["int", "splitter", "generator"]) +def test_fit_with_feature_importances(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) + cv = {"int": 3, "splitter": StratifiedKFold(n_splits=3)}.get(cv_type) + if cv_type == "generator": + cv = StratifiedKFold(n_splits=3).split(X, y) + + selector = BaseRecursiveSelector( + RandomForestClassifier(n_estimators=5, random_state=1), + scoring="roc_auc", + cv=cv, + variables=VARIABLES, + ) + selector.fit(X, y) + + assert selector.variables_ == VARIABLES + assert selector.feature_names_in_ == [f"var_{i}" for i in range(12)] + assert selector.n_features_in_ == 12 + assert selector.initial_model_performance_ == pytest.approx(0.9947395643178775) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.05016049596989658, + "var_4": 0.5704427408209541, + "var_7": 0.37939676320914945, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.02070511883333705, + "var_4": 0.031350384984021235, + "var_7": 0.03942210400299374, + } + ) + + +def test_fit_with_coefficients(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 2.0421118575507067, + "var_4": 0.36861802531867144, + "var_7": 2.68269483714549, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.08627973281236784, + "var_4": 0.12490483002792199, + "var_7": 0.13862536091514688, + } + ) + + +def test_fit_with_permutation_importance(make_df, data_classification): + # KNN has no coef_ or feature_importances_, so importance comes from + # permutation_importance. + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + KNeighborsClassifier(), scoring="accuracy", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.991997986009962) + assert dict(selector.feature_importances_) == pytest.approx( + { + "var_0": 0.0050000000000000044, + "var_4": 0.0753333333333333, + "var_7": 0.42766666666666664, + } + ) + assert dict(selector.feature_importances_std_) == pytest.approx( + { + "var_0": 0.0010000000000000009, + "var_4": 0.0005773502691896263, + "var_7": 0.0037859388972001223, + } + ) + + +def test_feature_importances_type(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(selector.feature_importances_, importance_type) + assert isinstance(selector.feature_importances_std_, importance_type) + assert list(selector.feature_importances_.keys()) == VARIABLES + + +def test_fit_returns_narwhals_frame_and_target(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + nw_X, y_ = selector.fit(X, y) + + assert isinstance(nw_X, nw.DataFrame) + assert nw_X.to_native() is X + assert isinstance(y_, type(y)) + + +def test_fit_finds_numerical_variables(make_df, data_classification): + X, y = _split_target( + make_df, + { + "var_0": data_classification["var_0"], + "cat": ["a", "b"] * 500, + "var_4": data_classification["var_4"], + "target": data_classification["target"], + }, + ) + selector = BaseRecursiveSelector(LogisticRegression(), scoring="roc_auc", cv=3) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert selector.feature_names_in_ == ["var_0", "cat", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +def test_fit_with_confirm_variables(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), + scoring="roc_auc", + cv=3, + variables=["var_0", "var_4", "Hola"], + confirm_variables=True, + ) + selector.fit(X, y) + + assert selector.variables_ == ["var_0", "var_4"] + assert list(selector.feature_importances_.keys()) == ["var_0", "var_4"] + + +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_fit_with_list_and_array_target(make_df, data_classification, target_type): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + assert selector.initial_model_performance_ == pytest.approx(0.996732746280939) + + +def test_error_if_only_one_variable(make_df, data_classification): + X, y = _split_target(make_df, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=["var_0"] + ) + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['var_0']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.fit(X, y) + + +def test_feature_importances_are_pandas_series_indexed_by_variables( + data_classification, +): + X, y = _split_target(pd.DataFrame, data_classification) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=VARIABLES + ) + selector.fit(X, y) + + pd.testing.assert_series_equal( + selector.feature_importances_, + pd.Series( + [2.0421118575507067, 0.36861802531867144, 2.68269483714549], + index=VARIABLES, + ), + ) + + +def test_fit_with_integer_column_names(data_classification): + X, y = _split_target(pd.DataFrame, data_classification) + X.columns = list(range(12)) + selector = BaseRecursiveSelector( + LogisticRegression(), scoring="roc_auc", cv=3, variables=[0, 4, 7] + ) + selector.fit(X, y) + + assert selector.variables_ == [0, 4, 7] + assert selector.feature_names_in_ == list(range(12)) + assert dict(selector.feature_importances_) == pytest.approx( + {0: 2.0421118575507067, 4: 0.36861802531867144, 7: 2.68269483714549} + ) diff --git a/tests/test_selection/test_base_selection_functions.py b/tests/test_selection/test_base_selection_functions.py index b2345a53e..0c4cb0e03 100644 --- a/tests/test_selection/test_base_selection_functions.py +++ b/tests/test_selection/test_base_selection_functions.py @@ -1,333 +1,364 @@ +from datetime import datetime + +import numpy as np import pandas as pd import pytest from sklearn.ensemble import RandomForestClassifier -from sklearn.model_selection import StratifiedKFold, GroupKFold - +from sklearn.linear_model import Lasso, LogisticRegression +from sklearn.model_selection import GroupKFold, StratifiedKFold from feature_engine.selection.base_selection_functions import ( _select_all_variables, _select_numerical_variables, find_correlated_features, find_feature_importance, + get_feature_importances, single_feature_performance, ) - - -@pytest.fixture -def df(): - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "date_range": pd.date_range("2020-02-24", periods=4, freq="min"), - "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], - } +from tests.backend_helpers import make_series + +DATA_VARTYPES = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "date_range": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "date_obj0": ["2020-02-24", "2020-02-25", "2020-02-26", "2020-02-27"], +} + +# a is uncorrelated, b is correlated with c and d, and c and d only with b. +DATA_CORR = { + "a": [1, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, 0], + "c": [0, 0, 1, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], +} + +EXPECTED_MEAN = { + "var_0": 0.5813469607144305, + "var_1": 0.5325152703164752, + "var_2": 0.5023573007759755, + "var_3": 0.47596844810700234, + "var_4": 0.9696712897767115, + "var_5": 0.5078009005719849, + "var_6": 0.966096275433625, + "var_7": 0.9918595739378872, + "var_8": 0.521667767752105, + "var_9": 0.9476311088509884, + "var_10": 0.4871054926777818, + "var_11": 0.5180029642379039, +} + +EXPECTED_STD = { + "var_0": 0.0035430274728173775, + "var_1": 0.0046697767238672565, + "var_2": 0.023714708852568194, + "var_3": 0.04219857610624132, + "var_4": 0.010364079344188424, + "var_5": 0.03203946151605523, + "var_6": 0.0063709642968091335, + "var_7": 0.0014579159677989356, + "var_8": 0.027570153897628277, + "var_9": 0.014363240810578251, + "var_10": 0.020283618255582142, + "var_11": 0.02707242215734807, +} + + +EXPECTED_IMPORTANCE_MEAN = { + "var_0": 0.008110472647428566, + "var_1": 0.004425867029009318, + "var_2": 0.0014110527658847542, + "var_3": 0.0, + "var_4": 0.09519163119147249, + "var_5": 0.005151538162222261, + "var_6": 0.06819196935501609, + "var_7": 0.7958920351591532, + "var_8": 0.005514122712728161, + "var_9": 0.006699116878609683, + "var_10": 0.0006441114577744834, + "var_11": 0.008768082640701015, +} + +EXPECTED_IMPORTANCE_STD = { + "var_0": 0.0044896289532760465, + "var_1": 0.004400023300047043, + "var_2": 0.0012779435856766451, + "var_3": 0.0, + "var_4": 0.15831114958148926, + "var_5": 0.005860881861755391, + "var_6": 0.0666073858478959, + "var_7": 0.12339701768095801, + "var_8": 0.0037045342320885986, + "var_9": 0.0016309720229483885, + "var_10": 0.0011156337706026607, + "var_11": 0.005458498614789974, +} + + +def _split_target(make_df, data): + X = make_df({k: v for k, v in data.items() if k != "target"}) + return X, make_series(make_df, data["target"]) + + +def _pearson(x, y): + return np.corrcoef(x, y)[0, 1] + + +@pytest.fixture(scope="module") +def data_with_groups(): + rng = np.random.default_rng(1) + data = {f"var_{i}": rng.normal(size=100).tolist() for i in range(1, 6)} + data["target"] = rng.integers(0, 100, size=100).tolist() + groups = np.repeat(np.arange(1, 11), 10) + rng.shuffle(groups) + return data, groups.tolist() + + +@pytest.mark.parametrize( + "variables, confirm_variables, exclude_datetime, expected", + [ + (None, False, False, list(DATA_VARTYPES)), + (None, False, True, ["Name", "City", "Age", "Marks"]), + (["Name", "Age"], False, True, ["Name", "Age"]), + (["Name", "Age", "Hola"], True, True, ["Name", "Age"]), + ], +) +def test_select_all_variables( + make_df, variables, confirm_variables, exclude_datetime, expected +): + variables_ = _select_all_variables( + make_df(DATA_VARTYPES), variables, confirm_variables, exclude_datetime ) - df["Name"] = df["Name"].astype("category") - return df + assert variables_ == expected -def test_select_all_variables(df): - # select all variables - assert ( - _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=False - ) - == df.columns.to_list() +@pytest.mark.parametrize( + "variables, confirm_variables, expected", + [ + (None, False, ["Age", "Marks"]), + (["Marks"], False, ["Marks"]), + (["Marks", "Hola"], True, ["Marks"]), + ], +) +def test_select_numerical_variables(make_df, variables, confirm_variables, expected): + variables_ = _select_numerical_variables( + make_df(DATA_VARTYPES), variables, confirm_variables ) + assert variables_ == expected - # select all variables except datetime - assert _select_all_variables( - df, variables=None, confirm_variables=False, exclude_datetime=True - ) == ["Name", "City", "Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Name", "City", "Age", "Marks"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=False, exclude_datetime=True - ) - == subset - ) - # select subset of variables, with confirm - subset = ["Name", "City", "Age", "Marks", "Hola"] - assert ( - _select_all_variables( - df, variables=subset, confirm_variables=True, exclude_datetime=True - ) - == subset[:-1] +@pytest.mark.parametrize( + "variables, expected", + [ + (["a", "b", "c", "d"], ([{"b", "c", "d"}], ["c", "d"], {"b": {"c", "d"}})), + (["a", "c", "b", "d"], ([{"c", "b"}], ["b"], {"c": {"b"}})), + ], +) +def test_find_correlated_features(make_df, variables, expected): + X = make_df( + { + "a": [1, -1, 0, 0, 0, 0], + "b": [0, 0, 1, -1, 1, -1], + "c": [0, 0, 1, -1, 0, 0], + "d": [0, 0, 0, 0, 1, -1], + } ) + assert find_correlated_features(X, variables, "pearson", 0.7) == expected -def test_select_numerical_variables(df): - # select all numerical variables - assert _select_numerical_variables( - df, - variables=None, - confirm_variables=False, - ) == ["Age", "Marks"] - - # select subset of variables, without confirm - subset = ["Marks"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=False, - ) - == subset - ) +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall", _pearson]) +def test_find_correlated_features_methods(make_df, method): + X = make_df(DATA_CORR) + groups, drop, dict_ = find_correlated_features(X, list(DATA_CORR), method, 0.5) + assert groups == [{"b", "c", "d"}] + assert drop == ["c", "d"] + assert dict_ == {"b": {"c", "d"}} - # select subset of variables, with confirm - subset = ["Marks", "Hola"] - assert ( - _select_numerical_variables( - df, - variables=subset, - confirm_variables=True, - ) - == subset[:-1] - ) +@pytest.mark.parametrize( + "method, expected", + [ + ("pearson", ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ("spearman", ([{"b", "d"}], ["d"], {"b": {"d"}})), + ("kendall", ([{"b", "d"}], ["d"], {"b": {"d"}})), + (_pearson, ([{"a", "c"}, {"b", "d"}], ["c", "d"], {"a": {"c"}, "b": {"d"}})), + ], +) +def test_find_correlated_features_skips_missing_values(make_df, method, expected): + # each pair of variables is compared on the rows where neither is missing. + data = { + "a": [None, -1, 0, 0, 0, 0, 2], + "b": [0, 0, 1, -1, 1, -1, None], + "c": [0, 0, None, -1, 0, 0, 1], + "d": [0, 0, 0, 0, 1, -1, 1], + } + X = make_df(data) + assert find_correlated_features(X, list(data), method, 0.6) == expected -def test_find_correlated_features(): - # given a correlation-threshold of 0.7 - # a is uncorrelated, - # b is collinear to c and d, - # c and d are collinear only to b. - X = pd.DataFrame() - X["a"] = [1, -1, 0, 0, 0, 0] - X["b"] = [0, 0, 1, -1, 1, -1] - X["c"] = [0, 0, 1, -1, 0, 0] - X["d"] = [0, 0, 0, 0, 1, -1] +@pytest.mark.parametrize("method", ["pearson", "spearman", "kendall"]) +def test_find_correlated_features_ignores_constant_variables(make_df, method): + X = make_df({**DATA_CORR, "e": [1] * 7}) groups, drop, dict_ = find_correlated_features( - X, variables=["a", "b", "c", "d"], method="pearson", threshold=0.7 + X, ["e", *DATA_CORR], method, 0.5 ) - assert groups == [{"b", "c", "d"}] assert drop == ["c", "d"] assert dict_ == {"b": {"c", "d"}} - groups, drop, dict_ = find_correlated_features( - X, variables=["a", "c", "b", "d"], method="pearson", threshold=0.7 - ) - assert groups == [{"c", "b"}] - assert drop == ["b"] - assert dict_ == {"c": {"b"}} +def test_find_correlated_features_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA_CORR.values())}) + groups, drop, dict_ = find_correlated_features(X, [0, 1, 2, 3], "pearson", 0.5) + assert groups == [{1, 2, 3}] + assert drop == [2, 3] + assert dict_ == {1: {2, 3}} -def test_single_feature_performance(df_test): - X, y = df_test +def test_single_feature_performance(make_df, data_classification): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() mean_, std_ = single_feature_performance( X=X, y=y, - variables=variables, + variables=list(EXPECTED_MEAN), estimator=rf, cv=3, scoring="roc_auc", ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - expected_std = { - "var_0": 0.0035430274728173775, - "var_1": 0.0046697767238672565, - "var_2": 0.023714708852568194, - "var_3": 0.04219857610624132, - "var_4": 0.010364079344188424, - "var_5": 0.03203946151605523, - "var_6": 0.0063709642968091335, - "var_7": 0.0014579159677989356, - "var_8": 0.027570153897628277, - "var_9": 0.014363240810578251, - "var_10": 0.020283618255582142, - "var_11": 0.02707242215734807, - } - assert mean_ == expected_mean - assert std_ == expected_std + assert mean_ == pytest.approx(EXPECTED_MEAN) + assert std_ == pytest.approx(EXPECTED_STD) -def test_single_feature_performance_cv_generator(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_single_feature_performance_with_cv_splitter( + make_df, data_classification, cv_type +): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() cv = StratifiedKFold(n_splits=3) - for cv_ in [cv, cv.split(X, y)]: - mean_, _ = single_feature_performance( - X=X, - y=y, - variables=variables, - estimator=rf, - cv=cv_, - scoring="roc_auc", - ) - - expected_mean = { - "var_0": 0.5813469607144305, - "var_1": 0.5325152703164752, - "var_2": 0.5023573007759755, - "var_3": 0.47596844810700234, - "var_4": 0.9696712897767115, - "var_5": 0.5078009005719849, - "var_6": 0.966096275433625, - "var_7": 0.9918595739378872, - "var_8": 0.521667767752105, - "var_9": 0.9476311088509884, - "var_10": 0.4871054926777818, - "var_11": 0.5180029642379039, - } - assert mean_ == expected_mean + if cv_type == "generator": + cv = cv.split(X, y) + + mean_, _ = single_feature_performance( + X=X, y=y, variables=list(EXPECTED_MEAN), estimator=rf, cv=cv, scoring="roc_auc" + ) + assert mean_ == pytest.approx(EXPECTED_MEAN) -def test_single_feature_performance_with_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +@pytest.mark.parametrize("target_type", [list, np.array]) +def test_single_feature_performance_with_list_and_array_target( + make_df, data_classification, target_type +): + X, _ = _split_target(make_df, data_classification) + y = target_type(data_classification["target"]) rf = RandomForestClassifier(n_estimators=5, random_state=1) - variables = X.columns.to_list() - scoring = "neg_mean_absolute_error" + + mean_, _ = single_feature_performance( + X=X, y=y, variables=["var_0", "var_4"], estimator=rf, cv=3, scoring="roc_auc" + ) + assert mean_ == pytest.approx( + {"var_0": EXPECTED_MEAN["var_0"], "var_4": EXPECTED_MEAN["var_4"]} + ) + + +def test_single_feature_performance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) + rf = RandomForestClassifier(n_estimators=5, random_state=1) + variables = ["var_1", "var_2", "var_3", "var_4", "var_5"] cv = GroupKFold(n_splits=3) - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = single_feature_performance( X=X, y=y, variables=variables, estimator=rf, cv=cv, - scoring=scoring, + scoring="neg_mean_absolute_error", groups=groups, ) - assert mean_ == expected_mean_ assert std_ == expected_std_ -def test_find_feature_importance(df_test): - X, y = df_test +@pytest.mark.parametrize("cv_type", ["splitter", "generator"]) +def test_find_feature_importance(make_df, data_classification, cv_type): + X, y = _split_target(make_df, data_classification) rf = RandomForestClassifier(n_estimators=3, random_state=3) cv = StratifiedKFold(n_splits=3) - scoring = "recall" - - expected_mean = pd.Series( - data=[0.01, 0.0, 0.0, 0.0, 0.1, 0.01, 0.07, 0.8, 0.01, 0.01, 0.0, 0.01], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) - expected_std = pd.Series( - data=[ - 0.0045, - 0.0044, - 0.0013, - 0.0, - 0.1583, - 0.0059, - 0.0666, - 0.1234, - 0.0037, - 0.0016, - 0.0011, - 0.0055, - ], - index=[ - "var_0", - "var_1", - "var_2", - "var_3", - "var_4", - "var_5", - "var_6", - "var_7", - "var_8", - "var_9", - "var_10", - "var_11", - ], - ) + if cv_type == "generator": + cv = cv.split(X, y) mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv, - scoring=scoring, + X=X, y=y, estimator=rf, cv=cv, scoring="recall" ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) - mean_, std_ = find_feature_importance( - X=X, - y=y, - estimator=rf, - cv=cv.split(X, y), - scoring=scoring, - ) - pd.testing.assert_series_equal(mean_.round(2), expected_mean) - pd.testing.assert_series_equal(std_.round(4), expected_std) + importance_type = pd.Series if make_df is pd.DataFrame else dict + assert isinstance(mean_, importance_type) + assert isinstance(std_, importance_type) + assert dict(mean_) == pytest.approx(EXPECTED_IMPORTANCE_MEAN) + assert dict(std_) == pytest.approx(EXPECTED_IMPORTANCE_STD) -def test_find_feature_importancewith_groups(df_test_with_groups): - X, y, groups = df_test_with_groups +def test_find_feature_importance_with_groups(make_df, data_with_groups): + data, groups = data_with_groups + X, y = _split_target(make_df, data) rf = RandomForestClassifier(n_estimators=3, random_state=1) cv = GroupKFold(n_splits=3) - scoring = "neg_mean_absolute_error" - cv_indices = cv.split(X=X, y=y, groups=groups) expected_mean_, expected_std_ = find_feature_importance( X=X, y=y, estimator=rf, - cv=cv_indices, - scoring=scoring, + cv=cv.split(X=X, y=y, groups=groups), + scoring="neg_mean_absolute_error", ) - mean_, std_ = find_feature_importance( X=X, y=y, estimator=rf, cv=cv, - scoring=scoring, - groups=groups + scoring="neg_mean_absolute_error", + groups=groups, ) + assert dict(mean_) == dict(expected_mean_) + assert dict(std_) == dict(expected_std_) - pd.testing.assert_series_equal(mean_, expected_mean_) - pd.testing.assert_series_equal(std_, expected_std_) + +def test_find_feature_importance_returns_series_indexed_by_columns(): + X = pd.DataFrame({"a": [1.0, 2, 3, 4, 5, 6], "b": [0.5, 0, 1, 1, 3, 2]}) + X.columns.name = "features" + y = pd.Series([1.0, 2, 3, 4, 5, 6]) + + mean_, std_ = find_feature_importance( + X=X, y=y, estimator=Lasso(alpha=0.01), cv=2, scoring="r2" + ) + assert list(mean_.index) == ["a", "b"] + assert mean_.index.name == "features" + assert std_.index.name == "features" + + +@pytest.mark.parametrize( + "estimator, expected", + [ + (LogisticRegression(), [0.728 ** (1 / 3), 1.0]), + (Lasso(), [0.5, 2.0]), + ], +) +def test_get_feature_importances(estimator, expected): + if isinstance(estimator, Lasso): + estimator.coef_ = np.array([-0.5, 2.0]) + else: + estimator.coef_ = np.array([[0.6, 0.0], [0.0, 1.0], [0.8, 0.0]]) + assert get_feature_importances(estimator) == pytest.approx(expected) diff --git a/tests/test_selection/test_base_selector.py b/tests/test_selection/test_base_selector.py index 54cc5bfc0..7ef31226b 100644 --- a/tests/test_selection/test_base_selector.py +++ b/tests/test_selection/test_base_selector.py @@ -1,55 +1,163 @@ +import re +from datetime import datetime + +import numpy as np +import pandas as pd import pytest -from pandas.testing import assert_frame_equal +from sklearn.exceptions import NotFittedError from feature_engine.selection.base_selector import BaseSelector +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", None], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, None, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], +} + + +# init parameters +@pytest.mark.parametrize("confirm_variables", [None, "hola", [True], 1, 0.5]) +def test_error_if_confirm_variables_not_bool(confirm_variables): + msg = ( + "confirm_variables takes only values True and False. " + f"Got {confirm_variables} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + BaseSelector(confirm_variables=confirm_variables) -@pytest.mark.parametrize("val", [None, "hola", [True]]) -def test_confirm_variables_in_init(val): - with pytest.raises(ValueError): - BaseSelector(confirm_variables=val) +@pytest.mark.parametrize("confirm_variables", [True, False]) +def test_init_param_assignment(confirm_variables): + selector = BaseSelector(confirm_variables=confirm_variables) + assert selector.confirm_variables is confirm_variables -class MockClass(BaseSelector): - def __init__(self, variables=None, confirm_variables=False): - self.variables = variables - self.confirm_variables = confirm_variables +# fit and transform +class MockSelector(BaseSelector): + def __init__(self, features_to_drop=("Name", "Marks")): + self.features_to_drop = features_to_drop + self.confirm_variables = False def fit(self, X, y=None): - self.features_to_drop_ = ["Name", "Marks"] + self.features_to_drop_ = list(self.features_to_drop) self._get_feature_names_in(X) return self -def test_transform_method(df_vartypes): - transformer = MockClass() - transformer.fit(df_vartypes) - Xt = transformer.transform(df_vartypes) +def test_transform_drops_features(make_df): + X = make_df(DATA) + Xt = MockSelector().fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "dob": DATA["dob"], + } + + +def test_transform_restores_train_column_order(make_df): + selector = MockSelector().fit(make_df(DATA)) + X = make_df({var: DATA[var] for var in ["dob", "Marks", "Age", "Name", "City"]}) + Xt = selector.transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == ["City", "Age", "dob"] + + +@pytest.mark.parametrize( + "features_to_drop, expected", + [ + ([], ["Name", "City", "Age", "Marks", "dob"]), + (["dob"], ["Name", "City", "Age", "Marks"]), + (["Age", "Name", "City", "dob"], ["Marks"]), + ], +) +def test_transform_returns_retained_features(make_df, features_to_drop, expected): + X = make_df(DATA) + Xt = MockSelector(features_to_drop).fit(X).transform(X) + + assert isinstance(Xt, make_df) + assert list(Xt.columns) == expected + + +def test_transform_does_not_modify_input(make_df): + X = make_df(DATA) + MockSelector().fit(X).transform(X) + assert frame_to_dict(X) == frame_to_dict(make_df(DATA)) + - # tests output of transform - assert_frame_equal(Xt, df_vartypes.drop(["Name", "Marks"], axis=1)) +def test_error_if_transform_df_has_different_number_of_columns(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "The number of columns in this dataset is different from the one used to " + "fit this transformer (when using the fit() method)." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector.transform(make_df({"Age": DATA["Age"], "Marks": DATA["Marks"]})) + + +def test_error_if_transform_before_fit(make_df): + msg = ( + "This MockSelector instance is not fitted yet. Call 'fit' with " + "appropriate arguments before using this estimator." + ) + with pytest.raises(NotFittedError, match=re.escape(msg)): + MockSelector().transform(make_df(DATA)) - # tests this line: X = X[self.feature_names_in_] - assert_frame_equal( - transformer.transform(df_vartypes[["City", "Age", "Name", "Marks", "dob"]]), - Xt, + +def test_error_if_transform_input_not_dataframe(make_df): + selector = MockSelector().fit(make_df(DATA)) + msg = ( + "X must be a dataframe from a library supported by narwhals " + "(e.g. pandas, polars, PyArrow). Got instead." ) - # test error when there is a df shape missmatch - with pytest.raises(ValueError): - assert transformer.transform(df_vartypes[["Age", "Marks"]]) + with pytest.raises(TypeError, match=re.escape(msg)): + selector.transform(np.ones((4, 5))) + + +def test_get_feature_names_in(make_df): + selector = MockSelector() + selector._get_feature_names_in(make_df(DATA)) + assert selector.feature_names_in_ == ["Name", "City", "Age", "Marks", "dob"] + assert selector.n_features_in_ == 5 + + +def test_get_support(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_support() == [False, True, True, False, True] + assert list(selector.get_support(indices=True)) == [1, 2, 4] + + +def test_get_feature_names_out(make_df): + selector = MockSelector().fit(make_df(DATA)) + assert selector.get_feature_names_out() == ["City", "Age", "dob"] + + +def test_check_variable_number(): + selector = MockSelector() + selector.variables_ = ["Age"] + msg = ( + "The selector needs at least 2 or more variables to select from. " + "Got only 1 variable: ['Age']." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + selector._check_variable_number() + +def test_transform_with_integer_column_names(): + X = pd.DataFrame({i: values for i, values in enumerate(DATA.values())}) + selector = MockSelector(features_to_drop=[0, 3]).fit(X) + Xt = selector.transform(X[[4, 3, 2, 1, 0]]) -def test_get_feature_names_in(df_vartypes): - tr = MockClass() - tr._get_feature_names_in(df_vartypes) - assert tr.n_features_in_ == df_vartypes.shape[1] - assert tr.feature_names_in_ == list(df_vartypes.columns) + assert selector.feature_names_in_ == [0, 1, 2, 3, 4] + pd.testing.assert_frame_equal(Xt, X[[1, 2, 4]]) -def test_get_support(df_vartypes): - tr = MockClass() - tr.fit(df_vartypes) - v_bool = [False, True, True, False, True] - v_ind = [1, 2, 4] - assert tr.get_support() == v_bool - assert list(tr.get_support(indices=True)) == v_ind +def test_transform_keeps_pandas_index(): + X = pd.DataFrame(DATA, index=[10, 11, 12, 13]) + Xt = MockSelector().fit(X).transform(X) + pd.testing.assert_frame_equal(Xt, X[["City", "Age", "dob"]]) diff --git a/tests/test_selection/test_drop_constant_features.py b/tests/test_selection/test_drop_constant_features.py index a89bc24d6..4fe2339b3 100644 --- a/tests/test_selection/test_drop_constant_features.py +++ b/tests/test_selection/test_drop_constant_features.py @@ -1,175 +1,314 @@ -import numpy as np +import re +from datetime import datetime + import pandas as pd import pytest from feature_engine.selection import DropConstantFeatures +from tests.backend_helpers import frame_to_dict + +DATA = { + "Name": ["tom", "nick", "krish", "jack"], + "City": ["London", "Manchester", "Liverpool", "Bristol"], + "Age": [20, 21, 19, 18], + "Marks": [0.9, 0.8, 0.7, 0.6], + "dob": [datetime(2020, 2, 24, 0, minute) for minute in range(4)], + "const_feat_num": [1, 1, 1, 1], + "const_feat_cat": ["a", "a", "a", "a"], + "quasi_feat_num": [1, 1, 1, 2], + "quasi_feat_cat": ["a", "a", "a", "b"], +} + +DATA_NA = { + "num": [1.0, 2.0, 3.0, 4.0, 5.0], + "const_num_na": [1.0, 1.0, 1.0, None, 1.0], + "const_cat_na": ["a", "a", None, "a", "a"], + "quasi_num_na": [1.0, 1.0, 1.0, None, None], + "quasi_cat_na": ["a", None, "b", None, None], + "cat": ["a", "b", "c", "d", "e"], +} + + +# init parameters +@pytest.mark.parametrize("tol", [2, -0.1, 1.5, "hola", False, None, [0.5]]) +def test_error_if_tol_not_allowed(tol): + msg = f"tol must be a float or integer between 0 and 1. Got {tol} instead." + with pytest.raises(ValueError, match=re.escape(msg)): + DropConstantFeatures(tol=tol) -@pytest.fixture(scope="module") -def df_constant_features(): - data = { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "const_feat_num": [1, 1, 1, 1], - "const_feat_cat": ["a", "a", "a", "a"], - "quasi_feat_num": [1, 1, 1, 2], - "quasi_feat_cat": ["a", "a", "a", "b"], - } +@pytest.mark.parametrize("missing_values", [2, "hola", False, None, ["raise"]]) +def test_error_if_missing_values_not_allowed(missing_values): + msg = ( + "missing_values takes only values 'raise', 'ignore' or 'include'. " + f"Got {missing_values} instead." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropConstantFeatures(missing_values=missing_values) - df = pd.DataFrame(data) - return df +@pytest.mark.parametrize( + "tol, missing_values, confirm_variables", + [(1, "raise", False), (0, "ignore", True), (0.7, "include", False)], +) +def test_init_param_assignment(tol, missing_values, confirm_variables): + sel = DropConstantFeatures( + tol=tol, missing_values=missing_values, confirm_variables=confirm_variables + ) + assert sel.tol == tol + assert sel.missing_values == missing_values + assert sel.confirm_variables is confirm_variables -def test_drop_constant_features(df_constant_features): - transformer = DropConstantFeatures(tol=1, variables=None) - X = transformer.fit_transform(df_constant_features) +# fit and transform +def test_drop_constant_features(make_df): + X = make_df(DATA) + sel = DropConstantFeatures() + Xt = sel.fit_transform(X) - # expected result - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "quasi_feat_num": [1, 1, 1, 2], - "quasi_feat_cat": ["a", "a", "a", "b"], - } - ) + assert sel.variables_ == list(DATA.keys()) + assert sel.features_to_drop_ == ["const_feat_num", "const_feat_cat"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + k: v for k, v in DATA.items() if k not in ["const_feat_num", "const_feat_cat"] + } - # fit attribute - assert transformer.features_to_drop_ == ["const_feat_num", "const_feat_cat"] - # transform output - pd.testing.assert_frame_equal(X, df) +@pytest.mark.parametrize( + "tol, expected", + [ + (0.8, ["const_feat_num", "const_feat_cat"]), + ( + 0.75, + ["const_feat_num", "const_feat_cat", "quasi_feat_num", "quasi_feat_cat"], + ), + ( + 0.7, + ["const_feat_num", "const_feat_cat", "quasi_feat_num", "quasi_feat_cat"], + ), + ], +) +def test_drop_quasi_constant_features(make_df, tol, expected): + X = make_df(DATA) + sel = DropConstantFeatures(tol=tol) + Xt = sel.fit_transform(X) + assert sel.features_to_drop_ == expected + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {k: v for k, v in DATA.items() if k not in expected} -def test_drop_constant_and_quasiconstant_features(df_constant_features): - transformer = DropConstantFeatures(tol=0.7, variables=None) - X = transformer.fit_transform(df_constant_features) - # expected result - df = pd.DataFrame( - { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - } +def test_variables_list(make_df): + X = make_df(DATA) + sel = DropConstantFeatures( + tol=0.7, variables=["Name", "const_feat_num", "quasi_feat_num"] ) + Xt = sel.fit_transform(X) - # fit attr - assert transformer.features_to_drop_ == [ - "const_feat_num", - "const_feat_cat", - "quasi_feat_num", - "quasi_feat_cat", - ] + assert sel.variables_ == ["Name", "const_feat_num", "quasi_feat_num"] + assert sel.features_to_drop_ == ["const_feat_num", "quasi_feat_num"] + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == { + k: v for k, v in DATA.items() if k not in ["const_feat_num", "quasi_feat_num"] + } - # transform params - pd.testing.assert_frame_equal(X, df) +def test_single_variable(make_df): + sel = DropConstantFeatures(variables="const_feat_cat").fit(make_df(DATA)) + assert sel.variables_ == ["const_feat_cat"] + assert sel.features_to_drop_ == ["const_feat_cat"] + + +def test_confirm_variables(make_df): + sel = DropConstantFeatures( + variables=["const_feat_num", "Age", "not_in_df"], confirm_variables=True + ).fit(make_df(DATA)) + assert sel.variables_ == ["const_feat_num", "Age"] + assert sel.features_to_drop_ == ["const_feat_num"] + + +@pytest.mark.parametrize( + "data, tol", + [ + ({"col1": [1, 1, 1], "col2": ["a", "a", "a"]}, 1), + ({"col1": [1, 1, 1, 1], "col2": [1, 1, 1, 2], "col3": [1, 2, 2, 2]}, 0.7), + ({"col1": [1, 2, 3, 4], "col2": [1, 1, 2, 2]}, 0), + ], +) +def test_error_if_all_features_are_dropped(make_df, data, tol): + msg = ( + "The resulting dataframe will have no columns after dropping all " + "constant or quasi-constant features. Try changing the tol value." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropConstantFeatures(tol=tol).fit(make_df(data)) -def test_drop_constant_features_with_list_of_variables(df_constant_features): - # test case 3: drop features showing threshold more than 0.7 with variable list - transformer = DropConstantFeatures( - tol=0.7, variables=["Name", "const_feat_num", "quasi_feat_num"] + +def test_error_if_missing_values_raise(make_df): + msg = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." ) - X = transformer.fit_transform(df_constant_features) + with pytest.raises(ValueError, match=re.escape(msg)): + DropConstantFeatures(missing_values="raise").fit(make_df(DATA_NA)) + + +def test_missing_values_raise_checks_only_selected_variables(make_df): + sel = DropConstantFeatures(variables=["num", "cat"], missing_values="raise") + sel.fit(make_df(DATA_NA)) + assert sel.features_to_drop_ == [] + + +@pytest.mark.parametrize( + "tol, expected", + [ + # with tol=1, variables with a single value besides missing data are dropped. + (1, ["const_num_na", "const_cat_na", "quasi_num_na"]), + (0.8, ["const_num_na", "const_cat_na"]), + (0.6, ["const_num_na", "const_cat_na", "quasi_num_na"]), + ], +) +def test_missing_values_ignore(make_df, tol, expected): + X = make_df(DATA_NA) + sel = DropConstantFeatures(tol=tol, missing_values="ignore") + Xt = sel.fit_transform(X) + + assert sel.features_to_drop_ == expected + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {k: v for k, v in DATA_NA.items() if k not in expected} + + +@pytest.mark.parametrize( + "tol, expected", + [ + (1, []), + (0.8, ["const_num_na", "const_cat_na"]), + (0.6, ["const_num_na", "const_cat_na", "quasi_num_na", "quasi_cat_na"]), + ], +) +def test_missing_values_include(make_df, tol, expected): + X = make_df(DATA_NA) + sel = DropConstantFeatures(tol=tol, missing_values="include") + Xt = sel.fit_transform(X) + + assert sel.features_to_drop_ == expected + assert isinstance(Xt, make_df) + assert frame_to_dict(Xt) == {k: v for k, v in DATA_NA.items() if k not in expected} + + +@pytest.mark.parametrize( + "missing_values, tol, expected", + [ + ("ignore", 1, []), + ("ignore", 0.5, []), + ("include", 1, ["all_na"]), + ("include", 0.5, ["all_na"]), + ], +) +def test_variable_with_only_missing_values(make_df, missing_values, tol, expected): + X = make_df({"all_na": [None, None, None], "num": [1.0, 2.0, 3.0]}) + sel = DropConstantFeatures(tol=tol, missing_values=missing_values).fit(X) + assert sel.features_to_drop_ == expected + + +@pytest.mark.parametrize( + "missing_values, tol, expected", + [ + ("ignore", 1, ["nan_num"]), + ("ignore", 0.5, ["nan_num"]), + ("ignore", 0.6, []), + ("include", 1, []), + ("include", 0.5, ["nan_num"]), + ], +) +def test_nan_is_treated_as_missing(make_df, missing_values, tol, expected): + nan = float("nan") + X = make_df({"nan_num": [nan, 1.0, nan, 1.0], "num": [1.0, 2.0, 3.0, 4.0]}) + sel = DropConstantFeatures(tol=tol, missing_values=missing_values).fit(X) + assert sel.features_to_drop_ == expected + + +def test_error_if_nan_and_missing_values_raise(make_df): + X = make_df({"nan_num": [float("nan"), 1.0, 2.0], "num": [1.0, 2.0, 3.0]}) + msg = ( + "Some of the variables in the dataset contain NaN. Check and " + "remove those before using this transformer." + ) + with pytest.raises(ValueError, match=re.escape(msg)): + DropConstantFeatures().fit(X) + - # expected result - df = pd.DataFrame( +@pytest.mark.parametrize("tol, expected", [(0.8, []), (0.4, ["literal"])]) +def test_include_counts_missing_values_apart_from_other_values(make_df, tol, expected): + X = make_df( { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "const_feat_cat": ["a", "a", "a", "a"], - "quasi_feat_cat": ["a", "a", "a", "b"], + "literal": ["missing_values", "missing_values", None, None, "x"], + "num": [1, 2, 3, 4, 5], } ) + sel = DropConstantFeatures(tol=tol, missing_values="include").fit(X) + assert sel.features_to_drop_ == expected - # fit attr - assert transformer.features_to_drop_ == ["const_feat_num", "quasi_feat_num"] - # transform params - pd.testing.assert_frame_equal(X, df) +def test_fit_does_not_modify_input(make_df): + X = make_df(DATA_NA) + DropConstantFeatures(tol=0.7, missing_values="include").fit_transform(X) + assert frame_to_dict(X) == DATA_NA -@pytest.mark.parametrize("tol", [2, "hola", False]) -def test_error_if_tol_value_not_allowed(tol): - # test case 5: threshold not between 0 and 1 - with pytest.raises(ValueError): - DropConstantFeatures(tol=tol) +def test_get_support(make_df): + sel = DropConstantFeatures(tol=0.7).fit(make_df(DATA)) + assert sel.get_support() == [True] * 5 + [False] * 4 -@pytest.mark.parametrize("tol", [1, 0, 0.5, 0.7]) -def test_tol_init_param(tol): - sel = DropConstantFeatures(tol=tol) - assert sel.tol == tol +def test_integer_column_names(): + X = pd.DataFrame({0: [1, 1, 1], 1: [1, 2, 3], "c": ["a", "a", "b"]}) + sel = DropConstantFeatures(tol=0.6) + Xt = sel.fit_transform(X) + assert sel.features_to_drop_ == [0, "c"] + pd.testing.assert_frame_equal(Xt, X[[1]]) -@pytest.mark.parametrize("missing", [2, "hola", False]) -def test_error_if_missing_values_not_permitted(missing): - # test case 5: threshold not between 0 and 1 - with pytest.raises(ValueError): - DropConstantFeatures(missing_values=missing) - - -def test_error_if_all_constant_and_quasi_constant_features(): - # test case 6: when input contains all constant features - with pytest.raises(ValueError): - DropConstantFeatures().fit(pd.DataFrame({"col1": [1, 1, 1], "col2": [1, 1, 1]})) - - # test case 7: when input contains all constant and quasi constant features - with pytest.raises(ValueError): - transformer = DropConstantFeatures(tol=0.7) - transformer.fit_transform( - pd.DataFrame( - { - "col1": [1, 1, 1, 1], - "col2": [1, 1, 1, 1], - "col3": [1, 1, 1, 2], - "col4": [1, 1, 1, 2], - } - ) - ) - - -def test_missing_values_param_functionality(): - - df = { - "Name": ["tom", "nick", "krish", "jack"], - "City": ["London", "Manchester", "Liverpool", "Bristol"], - "Age": [20, 21, 19, 18], - "Marks": [0.9, 0.8, 0.7, 0.6], - "dob": pd.date_range("2020-02-24", periods=4, freq="min"), - "const_feat_num": [1, 1, 1, np.nan], - "const_feat_cat": ["a", "a", "a", "a"], - "quasi_feat_num": [1, 1, 1, 2], - "quasi_feat_cat": ["a", "a", "a", np.nan], - } - df = pd.DataFrame(df) - - # test raises error if there is na - transformer = DropConstantFeatures(missing_values="raise") - with pytest.raises(ValueError): - transformer.fit(df) - - # test ignores na - transformer = DropConstantFeatures(missing_values="ignore").fit(df) - constant = ["const_feat_num", "const_feat_cat", "quasi_feat_cat"] - assert transformer.features_to_drop_ == constant - pd.testing.assert_frame_equal(df.drop(constant, axis=1), transformer.transform(df)) - - # test includes na - transformer = DropConstantFeatures(tol=0.7, missing_values="include").fit(df) - qconstant = ["const_feat_num", "const_feat_cat", "quasi_feat_num", "quasi_feat_cat"] - assert transformer.features_to_drop_ == qconstant - pd.testing.assert_frame_equal(df.drop(qconstant, axis=1), transformer.transform(df)) + +def test_pandas_index_is_preserved(): + X = pd.DataFrame({"a": [1, 1, 1, 1], "b": [1, 2, 3, 4]}, index=[10, 3, 7, 1]) + Xt = DropConstantFeatures().fit_transform(X) + pd.testing.assert_frame_equal(Xt, X[["b"]]) + + +@pytest.mark.parametrize( + "missing_values, tol, expected", + [ + ("ignore", 1, []), + ("ignore", 0.5, ["c"]), + ("include", 1, []), + ("include", 0.5, ["c"]), + ], +) +def test_pandas_category_dtype(missing_values, tol, expected): + # the unused category "z" must not count as a value. + X = pd.DataFrame( + { + "c": pd.Categorical(["a", "a", None, "b"], categories=["a", "b", "z"]), + "d": [1, 2, 3, 4], + } + ) + sel = DropConstantFeatures(tol=tol, missing_values=missing_values).fit(X) + assert sel.features_to_drop_ == expected + + +@pytest.mark.parametrize( + "missing_values, tol, expected", + [ + ("ignore", 1, ["a"]), + ("ignore", 0.75, ["a"]), + ("include", 1, []), + ("include", 0.75, ["a"]), + ], +) +def test_pandas_nullable_integer_dtype(missing_values, tol, expected): + X = pd.DataFrame( + {"a": pd.array([1, 1, None, 1], dtype="Int64"), "b": [1, 2, 3, 4]} + ) + sel = DropConstantFeatures(tol=tol, missing_values=missing_values).fit(X) + assert sel.features_to_drop_ == expected