Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
99 changes: 80 additions & 19 deletions docs/user_guide/selection/RecursiveFeatureAddition.rst
Original file line number Diff line number Diff line change
Expand Up @@ -136,7 +136,7 @@ entire dataset:

.. code:: python

0.488702767247119
np.float64(0.48870212980353145)

Evaluating feature importance
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Expand Down Expand Up @@ -202,15 +202,15 @@ In the following output we see the changes in performance returned by adding eac
.. code:: python

{'s1': 0,
's5': 0.28371458794131676,
'bmi': 0.1377714799388745,
's2': 0.0023327265047610735,
'bp': 0.018759914615172735,
'sex': 0.0027996354657459643,
's4': 0.002695149440021638,
's3': 0.002683934134630306,
's6': 0.000304067408860742,
'age': -0.007387230783454768}
's5': np.float64(0.28371458794131676),
'bmi': np.float64(0.13777147993887456),
's2': np.float64(0.0023327265047610735),
'bp': np.float64(0.01875991461517268),
'sex': np.float64(0.002799635465745798),
's4': np.float64(0.002695149440021638),
's3': np.float64(0.0026839341346303613),
's6': np.float64(0.0003040674088605755),
'age': np.float64(-0.007387230783454768)}

We can also check out the standard deviation of the performance drift:

Expand All @@ -225,15 +225,15 @@ returned by adding each feature:
.. code:: python

{'s1': 0,
's5': 0.029336910701570382,
'bmi': 0.01752426732750277,
's2': 0.020525965661877265,
'bp': 0.017326401244547558,
'sex': 0.00867675077259389,
's4': 0.024234566449074676,
's3': 0.023391851139598106,
's6': 0.016865740401721313,
'age': 0.02042081611218045}
's5': np.float64(0.02933691070157033),
'bmi': np.float64(0.017524267327502716),
's2': np.float64(0.020525965661877265),
'bp': np.float64(0.017326401244547592),
'sex': np.float64(0.008676750772593802),
's4': np.float64(0.024234566449074697),
's3': np.float64(0.02339185113959813),
's6': np.float64(0.01686574040172137),
'age': np.float64(0.020420816112180475)}

We can now plot the performance change with the standard deviation to identify importance
features:
Expand Down Expand Up @@ -321,6 +321,67 @@ be dropped:

[False, False, True, True, True, False, False, False, True, False]

With polars
~~~~~~~~~~~

:class:`RecursiveFeatureAddition` also selects features from polars dataframes, and
returns a polars dataframe. Let's load the diabetes dataset into a polars dataframe:

.. code:: python

import polars as pl

diabetes = load_diabetes()
X = pl.DataFrame(diabetes.data, schema=diabetes.feature_names)
y = pl.Series("target", diabetes.target)

Now, we select features as we did with pandas:

.. code:: python

tr = RecursiveFeatureAddition(estimator=LinearRegression(), scoring="r2", cv=3)
Xt = tr.fit_transform(X, y)
print(Xt.head())

The same 4 features are retained:

.. code:: text

shape: (5, 4)
┌───────────┬───────────┬───────────┬───────────┐
│ bmi ┆ bp ┆ s1 ┆ s5 │
│ --- ┆ --- ┆ --- ┆ --- │
│ f64 ┆ f64 ┆ f64 ┆ f64 │
╞═══════════╪═══════════╪═══════════╪═══════════╡
│ 0.061696 ┆ 0.021872 ┆ -0.044223 ┆ 0.019907 │
│ -0.051474 ┆ -0.026328 ┆ -0.008449 ┆ -0.068332 │
│ 0.044451 ┆ -0.00567 ┆ -0.045599 ┆ 0.002861 │
│ -0.011595 ┆ -0.036656 ┆ 0.012191 ┆ 0.022688 │
│ -0.036385 ┆ 0.021872 ┆ 0.003935 ┆ -0.031988 │
└───────────┴───────────┴───────────┴───────────┘

With polars, the feature importance and its standard deviation are dictionaries instead
of pandas Series:

.. code:: python

tr.feature_importances_

In the following output we see the features sorted by their importance:

.. code:: python

{'s1': 750.0238715216746,
's5': 741.4713367752565,
'bmi': 522.3301645404875,
's2': 436.67158399913274,
'bp': 322.0918016880965,
'sex': 238.6195264502995,
's4': 182.174833735295,
's3': 113.96599187843395,
's6': 64.76841724774016,
'age': 41.41804062408145}

And that's it! You now know how to select features by recursively adding them to a dataset.

Additional resources
Expand Down
7 changes: 5 additions & 2 deletions feature_engine/_docstrings/fit_attributes.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,11 +35,14 @@

# used by selection module
_feature_importances_docstring = """feature_importances_:
Pandas Series with the feature importance (comes from step 2)
The feature importance (comes from step 2). A pandas Series with the
features as index when X is a pandas dataframe, and a dictionary with the
features as keys otherwise.
""".rstrip()

_feature_importances_std_docstring = """feature_importances_std_:
Pandas Series with the standard deviation of the feature importance.
The standard deviation of the feature importance, as a pandas Series or a
dictionary, like `feature_importances_`.
""".rstrip()

_performance_drifts_docstring = """performance_drifts_:
Expand Down
96 changes: 48 additions & 48 deletions feature_engine/selection/base_recursive_selector.py
Original file line number Diff line number Diff line change
@@ -1,22 +1,23 @@
from types import GeneratorType
from typing import List, Union
from typing import List, Tuple, Union

import pandas as pd
import narwhals as nw
import numpy as np
from narwhals.typing import IntoDataFrame, IntoSeries
from sklearn.inspection import permutation_importance
from sklearn.model_selection import cross_validate

from feature_engine._check_init_parameters.check_variables import (
_check_variables_input_value,
)
from feature_engine.dataframe_checks import check_X_y
from feature_engine.selection.base_selection_functions import get_feature_importances
from feature_engine.selection.base_selection_functions import (
_importance_series,
_select_numerical_variables,
get_feature_importances,
)
from feature_engine.selection.base_selector import BaseSelector
from feature_engine.tags import _return_tags
from feature_engine.variable_handling import (
check_numerical_variables,
find_numerical_variables,
retain_variables_if_in_df,
)

Variables = Union[None, int, str, List[Union[str, int]]]

Expand Down Expand Up @@ -81,10 +82,13 @@ class BaseRecursiveSelector(BaseSelector):
Performance of the model trained using the original dataset.

feature_importances_:
Pandas Series with the feature importance (comes from step 2)
The feature importance (comes from step 2). A pandas Series with the
features as index when X is a pandas dataframe, and a dictionary with the
features as keys otherwise.

feature_importances_std_:
Pandas Series with the standard deviation of the feature importance.
The standard deviation of the feature importance, as a pandas Series or a
dictionary, like `feature_importances_`.

features_to_drop_:
List with the features to remove from the dataset.
Expand Down Expand Up @@ -116,7 +120,9 @@ def __init__(
):

if not isinstance(threshold, (int, float)):
raise ValueError("threshold can only be integer or float")
raise ValueError(
f"threshold must be an integer or a float. Got {threshold} instead."
)

super().__init__(confirm_variables)
self.variables = _check_variables_input_value(variables)
Expand All @@ -126,84 +132,78 @@ def __init__(
self.cv = cv
self.groups = groups

def fit(self, X: pd.DataFrame, y: pd.Series):
def fit(self, X: IntoDataFrame, y: IntoSeries) -> Tuple[nw.DataFrame, IntoSeries]:
"""
Find initial model performance. Sort features by importance.

Parameters
----------
X: pandas dataframe of shape = [n_samples, n_features]
X: dataframe of shape = [n_samples, n_features]
The input dataframe

y: array-like of shape (n_samples)
Target variable. Required to train the estimator.
"""

# check input dataframe
X, y = check_X_y(X, y)
Returns
-------
nw_X: narwhals dataframe
The input dataframe, as a narwhals dataframe.

if self.variables is None:
self.variables_ = find_numerical_variables(X)
else:
if self.confirm_variables is True:
variables_ = retain_variables_if_in_df(X, self.variables)
self.variables_ = check_numerical_variables(X, variables_)
else:
self.variables_ = check_numerical_variables(X, self.variables)
y: Series or numpy array
The target, checked.
"""
nw_X, y = check_X_y(X, y)

self.variables_ = _select_numerical_variables(
X, self.variables, self.confirm_variables
)

self._cv = list(self.cv) if isinstance(self.cv, GeneratorType) else self.cv

# check that there are more than 1 variable to select from
self._check_variable_number()

# save input features
self._get_feature_names_in(X)

X_model = nw_X.select(nw.col(*self.variables_)).to_native()

# train model with all features and cross-validation
model = cross_validate(
estimator=self.estimator,
X=X[self.variables_],
X=X_model,
y=y,
cv=self._cv,
groups=self.groups,
scoring=self.scoring,
return_estimator=True,
)

# store initial model performance
self.initial_model_performance_ = model["test_score"].mean()

# Initialize a dataframe that will contain the list of the feature/coeff
# importance for each cross validation fold
feature_importances_cv = pd.DataFrame()

# Populate the feature_importances_cv dataframe with columns containing
# the feature importance values for each model returned by the cross
# validation.
# There are as many columns as folds.
for i in range(len(model["estimator"])):
m = model["estimator"][i]

# one row of feature importance per cross-validation fold
importances = []
for m in model["estimator"]:
if hasattr(m, "feature_importances_") or hasattr(m, "coef_"):
feature_importances_cv[i] = get_feature_importances(m)
importances.append(get_feature_importances(m))
else:
r = permutation_importance(
m,
X[self.variables_],
X_model,
y,
n_repeats=1,
random_state=10,
)
feature_importances_cv[i] = r.importances_mean
importances.append(r.importances_mean)
importances_arr = np.array(importances)

# Add the variables as index to feature_importances_cv
feature_importances_cv.index = self.variables_

# Aggregate the feature importance returned in each fold
self.feature_importances_ = feature_importances_cv.mean(axis=1)
self.feature_importances_std_ = feature_importances_cv.std(axis=1)
self.feature_importances_ = _importance_series(
X, self.variables_, importances_arr.mean(axis=0)
)
self.feature_importances_std_ = _importance_series(
X, self.variables_, importances_arr.std(axis=0, ddof=1)
)

return X, y
return nw_X, y

def _more_tags(self):
tags_dict = _return_tags()
Expand Down
Loading