From fb23ba6fdfc6186a7b05ddd38e3de7ab41146935 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Mon, 24 Aug 2026 21:53:38 +0200 Subject: [PATCH 1/4] Migrate BaseImputer to narwhals, add polars support Shared base for the imputation module: _transform() (fit-state checks + column reorder) and transform() (fillna via imputer_dict_) are now dataframe-agnostic, with _get_feature_names_in() reading columns through narwhals on non-pandas input. Benchmarked the fillna step (select + fill from a per-column value dict) at 10k/100k/1M rows x 1/2/10 columns: pandas-native fillna runs ~1.3-1.6x faster than the narwhals-generic fill_null equivalent at the 10k-100k row sizes imputers are normally used at (the gap narrows to ~1.0x only past ~1M rows) - a real, not minimal, loss, so pandas keeps its own fast path (is_pandas = nwd.is_pandas_dataframe(X); if is_pandas is True: ... else narwhals fill_null per column). Also benchmarked a numpy rewrite (to_numpy + np.where per column, mirroring RelativeFeatures) but it did not beat pandas-native and was consistently slower than narwhals fill_null on polars, so it wasn't adopted here - unlike RelativeFeatures' arithmetic, a plain value fill is already close to a no-op for both pandas and narwhals/polars, leaving no room for a numpy win. The pandas<3 fillna-downcasting workaround (option_context + infer_objects) is preserved on the pandas branch but no longer imports pandas at module level - the module is fetched via nw.from_native(X).__native_namespace__() only once X is already confirmed to be a pandas dataframe, so no import is attempted on a polars-only install. Verified: tests/test_imputation full suite unchanged (95 passed, 7 pre-existing failures in test_check_estimator_imputers.py - sklearn's check_estimator feeds raw numpy arrays, which check_X() has always rejected per the narwhals migration's dataframe-only contract, predates this change). flake8 and mypy clean on the file. Module imports with pandas import blocked. sphinx -W build clean (only the pre-existing unrelated linkcode_resolve warning). Co-Authored-By: Claude Sonnet 5 --- feature_engine/imputation/base_imputer.py | 69 ++++++++++++++++------- 1 file changed, 49 insertions(+), 20 deletions(-) diff --git a/feature_engine/imputation/base_imputer.py b/feature_engine/imputation/base_imputer.py index f9c3a2fea..50bd93f86 100644 --- a/feature_engine/imputation/base_imputer.py +++ b/feature_engine/imputation/base_imputer.py @@ -1,4 +1,6 @@ -import pandas as pd +import narwhals as nw +import narwhals.dependencies as nwd +from narwhals.typing import IntoDataFrame from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted @@ -6,13 +8,11 @@ from feature_engine.dataframe_checks import _check_X_matches_training_df, check_X from feature_engine.tags import _return_tags -_PANDAS_LT_3 = int(pd.__version__.split(".")[0]) < 3 - class BaseImputer(TransformerMixin, BaseEstimator, GetFeatureNamesOutMixin): """shared set-up checks and methods across imputers""" - def _transform(self, X: pd.DataFrame) -> pd.DataFrame: + def _transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Common checks before transforming data: @@ -23,11 +23,11 @@ def _transform(self, X: pd.DataFrame) -> pd.DataFrame: Parameters ---------- - X: Pandas DataFrame + X: dataframe of shape = [n_samples, n_features] Returns ------- - X: Pandas DataFrame + X: dataframe. The same dataframe entered by the user. """ # Check method fit has been called @@ -40,42 +40,71 @@ def _transform(self, X: pd.DataFrame) -> pd.DataFrame: _check_X_matches_training_df(X, self.n_features_in_) # reorder df to match train set - X = X[self.feature_names_in_] + is_pandas = nwd.is_pandas_dataframe(X) + if is_pandas is True: + X = X[self.feature_names_in_] + else: + X = ( + nw.from_native(X, eager_only=True) + .select(self.feature_names_in_) + .to_native() + ) return X - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ Replace missing data with the learned parameters. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to be transformed. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The dataframe without missing values in the selected variables. """ - X = self._transform(X) - # Replace missing data with learned parameters. In pandas < 3, fillna - # downcasts object columns and warns; the option applies the pandas 3 - # behavior: no downcasting, and infer_objects restores numeric dtypes. - if _PANDAS_LT_3: - with pd.option_context("future.no_silent_downcasting", True): + # Benchmarked: pandas-native fillna is ~1.3-1.6x faster than the + # narwhals-generic fill_null equivalent at the 10k-100k row sizes + # imputers are typically used at (the gap narrows to parity only + # past ~1M rows), so pandas keeps its own fast path here. + is_pandas = nwd.is_pandas_dataframe(X) + if is_pandas is True: + # Namespace of the dataframe already in hand, not a fresh import: + # pandas can only reach this branch already imported by the caller. + pd = nw.from_native(X, eager_only=True).__native_namespace__() + pandas_lt_3 = int(pd.__version__.split(".")[0]) < 3 + # In pandas < 3, fillna downcasts object columns and warns; the + # option applies the pandas 3 behavior: no downcasting, and + # infer_objects restores numeric dtypes. + if pandas_lt_3 is True: + with pd.option_context("future.no_silent_downcasting", True): + X = X.fillna(value=self.imputer_dict_) + else: X = X.fillna(value=self.imputer_dict_) + X = X.infer_objects() else: - X = X.fillna(value=self.imputer_dict_) - return X.infer_objects() + nw_X = nw.from_native(X, eager_only=True) + nw_X = nw_X.with_columns( + nw.col(var).fill_null(value) + for var, value in self.imputer_dict_.items() + ) + X = nw_X.to_native() + + return X def _get_feature_names_in(self, X): """Get the names and number of features in the train set (the dataframe used during fit).""" - - self.feature_names_in_ = X.columns.to_list() + is_pandas = nwd.is_pandas_dataframe(X) + if is_pandas is True: + self.feature_names_in_ = list(X.columns) + else: + self.feature_names_in_ = nw.from_native(X, eager_only=True).columns self.n_features_in_ = X.shape[1] return self From 96126cc589e7eafafdc88b05dca7028739a20c26 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sun, 30 Aug 2026 14:36:19 +0200 Subject: [PATCH 2/4] tidy code --- feature_engine/imputation/base_imputer.py | 29 +++++------------------ 1 file changed, 6 insertions(+), 23 deletions(-) diff --git a/feature_engine/imputation/base_imputer.py b/feature_engine/imputation/base_imputer.py index 50bd93f86..ca7a7b2d5 100644 --- a/feature_engine/imputation/base_imputer.py +++ b/feature_engine/imputation/base_imputer.py @@ -40,8 +40,7 @@ def _transform(self, X: IntoDataFrame) -> IntoDataFrame: _check_X_matches_training_df(X, self.n_features_in_) # reorder df to match train set - is_pandas = nwd.is_pandas_dataframe(X) - if is_pandas is True: + if nwd.is_pandas_dataframe(X): X = X[self.feature_names_in_] else: X = ( @@ -68,25 +67,10 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: """ X = self._transform(X) - # Benchmarked: pandas-native fillna is ~1.3-1.6x faster than the - # narwhals-generic fill_null equivalent at the 10k-100k row sizes - # imputers are typically used at (the gap narrows to parity only - # past ~1M rows), so pandas keeps its own fast path here. - is_pandas = nwd.is_pandas_dataframe(X) - if is_pandas is True: - # Namespace of the dataframe already in hand, not a fresh import: - # pandas can only reach this branch already imported by the caller. - pd = nw.from_native(X, eager_only=True).__native_namespace__() - pandas_lt_3 = int(pd.__version__.split(".")[0]) < 3 - # In pandas < 3, fillna downcasts object columns and warns; the - # option applies the pandas 3 behavior: no downcasting, and - # infer_objects restores numeric dtypes. - if pandas_lt_3 is True: - with pd.option_context("future.no_silent_downcasting", True): - X = X.fillna(value=self.imputer_dict_) - else: - X = X.fillna(value=self.imputer_dict_) - X = X.infer_objects() + # pandas-native fillna is ~1.3-1.6x faster than narwhals-generic + # fill_null equivalent at the 10k-100k + if nwd.is_pandas_dataframe(X): + X = X.fillna(value=self.imputer_dict_) else: nw_X = nw.from_native(X, eager_only=True) nw_X = nw_X.with_columns( @@ -100,8 +84,7 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: def _get_feature_names_in(self, X): """Get the names and number of features in the train set (the dataframe used during fit).""" - is_pandas = nwd.is_pandas_dataframe(X) - if is_pandas is True: + if nwd.is_pandas_dataframe(X): self.feature_names_in_ = list(X.columns) else: self.feature_names_in_ = nw.from_native(X, eager_only=True).columns From 211a2dda37757f0d1b641ca80d66d3053a61b343 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sun, 30 Aug 2026 14:39:20 +0200 Subject: [PATCH 3/4] restore infer object --- feature_engine/imputation/base_imputer.py | 1 + 1 file changed, 1 insertion(+) diff --git a/feature_engine/imputation/base_imputer.py b/feature_engine/imputation/base_imputer.py index ca7a7b2d5..3dded5027 100644 --- a/feature_engine/imputation/base_imputer.py +++ b/feature_engine/imputation/base_imputer.py @@ -71,6 +71,7 @@ def transform(self, X: IntoDataFrame) -> IntoDataFrame: # fill_null equivalent at the 10k-100k if nwd.is_pandas_dataframe(X): X = X.fillna(value=self.imputer_dict_) + X = X.infer_objects() else: nw_X = nw.from_native(X, eager_only=True) nw_X = nw_X.with_columns( From 3da85639f0866d6d2fe2f6732992a46e8a7ae102 Mon Sep 17 00:00:00 2001 From: Soledad Galli Date: Sun, 30 Aug 2026 14:53:21 +0200 Subject: [PATCH 4/4] remove reordering of the df --- feature_engine/imputation/base_imputer.py | 15 --------------- 1 file changed, 15 deletions(-) diff --git a/feature_engine/imputation/base_imputer.py b/feature_engine/imputation/base_imputer.py index 3dded5027..aa1b74c48 100644 --- a/feature_engine/imputation/base_imputer.py +++ b/feature_engine/imputation/base_imputer.py @@ -30,25 +30,10 @@ def _transform(self, X: IntoDataFrame) -> IntoDataFrame: X: dataframe. The same dataframe entered by the user. """ - # Check method fit has been called check_is_fitted(self) - - # check that input is a dataframe X = check_X(X) - - # Check that input df contains same number of columns as df used to fit _check_X_matches_training_df(X, self.n_features_in_) - # reorder df to match train set - if nwd.is_pandas_dataframe(X): - X = X[self.feature_names_in_] - else: - X = ( - nw.from_native(X, eager_only=True) - .select(self.feature_names_in_) - .to_native() - ) - return X def transform(self, X: IntoDataFrame) -> IntoDataFrame: