diff --git a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst index 74e150763..940746d9d 100644 --- a/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst +++ b/docs/user_guide/discretisation/GeometricWidthDiscretiser.rst @@ -144,6 +144,66 @@ In the following output, we see the interval limits determined for each variable 2212.974, inf]} +With polars +----------- + +:class:`GeometricWidthDiscretiser()` works in the same way with a polars dataframe: + +.. code:: python + + import numpy as np + import polars as pl + from feature_engine.discretisation import GeometricWidthDiscretiser + + np.random.seed(42) + df = pl.DataFrame({"x": np.random.randint(1, 100, 100).astype(float)}) + + disc = GeometricWidthDiscretiser(bins=10) + Xt = disc.fit_transform(df) + + print(Xt["x"].value_counts().sort("x")) + +The resulting bin counts: + +.. code:: text + + shape: (9, 2) + ┌─────┬───────┐ + │ x ┆ count │ + │ --- ┆ --- │ + │ i64 ┆ u32 │ + ╞═════╪═══════╡ + │ 0 ┆ 6 │ + │ 1 ┆ 3 │ + │ 3 ┆ 3 │ + │ 4 ┆ 1 │ + │ 5 ┆ 5 │ + │ 6 ┆ 9 │ + │ 7 ┆ 8 │ + │ 8 ┆ 25 │ + │ 9 ┆ 40 │ + └─────┴───────┘ + +And the fitted bin edges, matching what we'd get fitting on the same values with pandas: + +.. code:: python + + disc.binner_dict_ + +.. code:: python + + {'x': [-inf, + 3.573433146226546, + 4.475691865644366, + 5.895335641248283, + 8.129050213617685, + 11.643650760992958, + 17.173639757979174, + 25.874707744105372, + 39.565256521047, + 61.106419756718246, + inf]} + Interval width ~~~~~~~~~~~~~~ diff --git a/feature_engine/discretisation/base_discretiser.py b/feature_engine/discretisation/base_discretiser.py index 6c61d05d3..8bce3021f 100644 --- a/feature_engine/discretisation/base_discretiser.py +++ b/feature_engine/discretisation/base_discretiser.py @@ -1,7 +1,11 @@ # Authors: Morgan Sell # License: BSD 3 clause -import pandas as pd +from typing import List + +import narwhals as nw +import numpy as np +from narwhals.typing import IntoDataFrame from feature_engine._base_transformers.base_numerical import BaseNumericalTransformer @@ -41,45 +45,133 @@ def __init__( self.return_boundaries = return_boundaries self.precision = precision - def transform(self, X: pd.DataFrame) -> pd.DataFrame: + def transform(self, X: IntoDataFrame) -> IntoDataFrame: """Sort the variable values into the intervals. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The data to transform. Returns ------- - X_new: pandas dataframe of shape = [n_samples, n_features] + X_new: dataframe of shape = [n_samples, n_features] The transformed data with the discrete variables. """ # check input dataframe and if class was fitted X = self._check_transform_input_and_state(X) - # transform variables + # bin edges are already fixed by fit(), so sorting values into them is a + # plain numpy searchsorted - vectorizable identically for every backend, + # no pandas/polars-specific path needed. + nw_X = nw.from_native(X, eager_only=True) + native_namespace = nw_X.__native_namespace__() + if self.return_boundaries is True: - for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - precision=self.precision, - include_lowest=True, + new_columns = [ + nw.new_series( + feature, + _bin_labels( + nw_X.get_column(feature).to_numpy(), + self.binner_dict_[feature], + self.precision, + ), + backend=native_namespace, ) - X[self.variables_] = X[self.variables_].astype(str) - + for feature in self.variables_ + ] else: - for feature in self.variables_: - X[feature] = pd.cut( - X[feature], - self.binner_dict_[feature], - labels=False, - include_lowest=True, + # nw.Object mirrors the pandas "O" dtype astype() used to produce, + # and is what feature-engine's categorical encoders detect on + # every narwhals-supported backend (see variable_handling). + dtype = nw.Object if self.return_object is True else None + new_columns = [ + nw.new_series( + feature, + _bin_codes( + nw_X.get_column(feature).to_numpy(), + self.binner_dict_[feature], + self.return_object, + ), + dtype=dtype, + backend=native_namespace, ) + for feature in self.variables_ + ] - # return object - if self.return_object: - X[self.variables_] = X[self.variables_].astype("O") + X = nw_X.with_columns(*new_columns).to_native() return X + + +def _digitize(values: np.ndarray, bins_arr: np.ndarray): + """0-based bin index per value, right-closed intervals with the lowest edge + included - mirrors pandas.cut(bins=bins, include_lowest=True), which is + itself built on this same bins.searchsorted() call. Values outside the + bin range, and NaNs, are flagged via na_mask rather than given a code. + """ + ids = np.asarray(np.searchsorted(bins_arr, values, side="left")) + ids[values == bins_arr[0]] = 1 + na_mask: np.ndarray = np.isnan(values) | (ids == len(bins_arr)) | (ids == 0) + return ids - 1, na_mask + + +def _bin_codes(values: np.ndarray, bins: List[float], return_object: bool): + bins_arr: np.ndarray = np.asarray(bins, dtype=float) + codes, na_mask = _digitize(values, bins_arr) + + # match pandas.cut(labels=False): int codes, upcast to float only when a + # NaN placeholder is actually needed. + if na_mask.any(): + codes = codes.astype(np.float64) + codes[na_mask] = np.nan + if return_object is True: + codes = codes.astype(object) + + return codes + + +def _bin_labels(values: np.ndarray, bins: List[float], precision: int): + bins_arr: np.ndarray = np.asarray(bins, dtype=float) + codes, na_mask = _digitize(values, bins_arr) + + labels = np.asarray(_format_bin_labels(bins_arr, precision), dtype=object) + out: np.ndarray = np.empty(len(values), dtype=object) + out[~na_mask] = labels[codes[~na_mask]] + out[na_mask] = None + + return out + + +def _format_bin_labels(bins_arr: np.ndarray, precision: int) -> List[str]: + """"(lower, upper]" text per bin, replicating pandas.cut's own label + formatting: widen precision until break values are unique, then shrink + the lowest edge so include_lowest values still read as inside the first + interval. + """ + precision = _infer_precision(precision, bins_arr) + breaks = [_round_frac(b, precision) for b in bins_arr] + breaks[0] = breaks[0] - 10 ** (-precision) + return [f"({breaks[i]}, {breaks[i + 1]}]" for i in range(len(breaks) - 1)] + + +def _round_frac(x: float, precision: int) -> float: + if not np.isfinite(x) or x == 0: + return float(x) + frac, whole = np.modf(x) + if whole == 0: + digits = -int(np.floor(np.log10(abs(frac)))) - 1 + precision + else: + digits = precision + return float(np.around(x, digits)) + + +def _infer_precision(base_precision: int, bins_arr: np.ndarray) -> int: + # widen precision until every rounded break is unique - otherwise two + # adjacent bins could render with identical label text. + for precision in range(base_precision, 20): + levels = [_round_frac(b, precision) for b in bins_arr] + if len(set(levels)) == len(bins_arr): + return precision + return base_precision diff --git a/feature_engine/discretisation/geometric_width.py b/feature_engine/discretisation/geometric_width.py index 709381c71..41aa9bd18 100644 --- a/feature_engine/discretisation/geometric_width.py +++ b/feature_engine/discretisation/geometric_width.py @@ -1,7 +1,8 @@ from typing import List, Optional, Union +import narwhals as nw import numpy as np -import pandas as pd +from narwhals.typing import IntoDataFrame, IntoSeries from feature_engine._check_init_parameters.check_init_input_params import ( _check_return_empty_is_bool, @@ -159,14 +160,14 @@ def __init__( self.return_empty = return_empty self.bins = bins - def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): + def fit(self, X: IntoDataFrame, y: Optional[IntoSeries] = None): """ Learn the boundaries of the geometric width intervals / bins for each variable. Parameters ---------- - X: pandas dataframe of shape = [n_samples, n_features] + X: dataframe of shape = [n_samples, n_features] The training dataset. Can be the entire dataframe, not just the variables to be transformed. y: None @@ -177,10 +178,12 @@ def fit(self, X: pd.DataFrame, y: Optional[pd.Series] = None): X, variables_ = self._fit_setup(X) # fit + nw_X = nw.from_native(X, eager_only=True) binner_dict_ = {} for var in variables_: - min_, max_ = X[var].min(), X[var].max() + col = nw_X.get_column(var) + min_, max_ = col.min(), col.max() increment = np.power(max_ - min_, 1.0 / self.bins) bins = np.r_[ -np.inf, min_ + np.power(increment, np.arange(1, self.bins)), np.inf diff --git a/tests/test_discretisation/test_base_discretizer.py b/tests/test_discretisation/test_base_discretizer.py index fc8110ff1..f852bf164 100644 --- a/tests/test_discretisation/test_base_discretizer.py +++ b/tests/test_discretisation/test_base_discretizer.py @@ -1,5 +1,6 @@ import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.datasets import fetch_california_housing @@ -38,42 +39,42 @@ def test_correct_param_assignment_at_init(params): class MockClassFit(BaseDiscretiser): def fit(self, X): - california_dataset = fetch_california_housing() - data = pd.DataFrame( - california_dataset.data, columns=california_dataset.feature_names - ) + # bins are hard-coded rather than learnt, so this mock works unchanged + # on both pandas and polars input. self.variables_ = ["HouseAge"] self.binner_dict_ = {"HouseAge": [0, 20, 40, 60, np.inf]} - self.n_features_in_ = data.shape[1] - self.feature_names_in_ = california_dataset.feature_names + self.n_features_in_ = X.shape[1] + self.feature_names_in_ = list(X.columns) return self -def test_transform(): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_transform(make_df): california_dataset = fetch_california_housing() - data = pd.DataFrame( + data_pd = pd.DataFrame( california_dataset.data, columns=california_dataset.feature_names ) - data_t1 = data.copy() - data_t2 = data.copy() - - # HouseAge is the median house age in the block group. - data_t1["HouseAge"] = pd.cut( - data["HouseAge"], bins=[0, 20, 40, 60, np.inf], include_lowest=True - ) - data_t1["HouseAge"] = data_t1["HouseAge"].astype(str) - data_t2["HouseAge"] = pd.cut( - data["HouseAge"], + # ground truth via pandas.cut: bins are fixed by MockClassFit, so both + # backends must reproduce this exact output. + expected_codes = pd.cut( + data_pd["HouseAge"], bins=[0, 20, 40, 60, np.inf], labels=False, include_lowest=True, + ).to_numpy() + expected_labels = ( + pd.cut(data_pd["HouseAge"], bins=[0, 20, 40, 60, np.inf], include_lowest=True) + .astype(str) + .to_numpy() ) + data = make_df(data_pd) + transformer = MockClassFit(return_boundaries=False) X = transformer.fit_transform(data) - pd.testing.assert_frame_equal(X, data_t2) + assert np.array_equal(X["HouseAge"].to_numpy(), expected_codes) transformer = MockClassFit(return_object=False, return_boundaries=True) X = transformer.fit_transform(data) - pd.testing.assert_frame_equal(X, data_t1) + assert np.array_equal(X["HouseAge"].to_numpy(), expected_labels) diff --git a/tests/test_discretisation/test_geometric_width_discretiser.py b/tests/test_discretisation/test_geometric_width_discretiser.py index 6a4b56c2d..3e2e1c43a 100644 --- a/tests/test_discretisation/test_geometric_width_discretiser.py +++ b/tests/test_discretisation/test_geometric_width_discretiser.py @@ -1,11 +1,27 @@ +import narwhals as nw import numpy as np import pandas as pd +import polars as pl import pytest from sklearn.exceptions import NotFittedError from feature_engine.discretisation import GeometricWidthDiscretiser +def _normal_dist_data(): + np.random.seed(0) + mu, sigma = 0, 0.1 # mean and standard deviation + return {"var": list(np.random.normal(mu, sigma, 100))} + + +def _get_column_values(X, column): + return nw.from_native(X, eager_only=True).get_column(column).to_list() + + +def _get_column_dtype(X, column): + return nw.from_native(X, eager_only=True).get_column(column).dtype + + # test init params @pytest.mark.parametrize("param", [0.1, "hola", (True, False), {"a": True}, 2]) def test_raises_error_when_return_object_not_bool(param): @@ -43,14 +59,19 @@ def test_correct_param_assignment_at_init(params): assert t.bins == param2 -def test_fit_and_transform_methods(df_normal_dist): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_fit_and_transform_methods(make_df): + data = _normal_dist_data() + df = make_df(data) + transformer = GeometricWidthDiscretiser( bins=10, variables=None, return_object=False ) - X = transformer.fit_transform(df_normal_dist) + X = transformer.fit_transform(df) # manual calculation - min_, max_ = df_normal_dist["var"].min(), df_normal_dist["var"].max() + arr = np.array(data["var"]) + min_, max_ = arr.min(), arr.max() increment = np.power(max_ - min_, 1.0 / 10) bins = np.r_[-np.inf, min_ + np.power(increment, np.arange(1, 10)), np.inf] bins = np.sort(bins) @@ -58,34 +79,42 @@ def test_fit_and_transform_methods(df_normal_dist): # fit params assert (transformer.binner_dict_["var"] == bins).all() - # transform params - assert ( - X["var"] == pd.cut(df_normal_dist["var"], bins=bins, precision=7).cat.codes - ).all() + # transform params - ground truth from pandas.cut on the same bins; values + # must match regardless of which backend the input dataframe uses. + expected = list(pd.cut(pd.Series(arr), bins=bins, precision=7).cat.codes) + assert _get_column_values(X, "var") == expected -def test_automatically_find_variables_and_return_as_object(df_normal_dist): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_automatically_find_variables_and_return_as_object(make_df): + df = make_df(_normal_dist_data()) transformer = GeometricWidthDiscretiser(bins=10, variables=None, return_object=True) - X = transformer.fit_transform(df_normal_dist) - assert X["var"].dtypes == "O" + X = transformer.fit_transform(df) + assert _get_column_dtype(X, "var") == nw.Object -def test_error_if_input_df_contains_na_in_fit(df_na): - # test case 3: when dataset contains na, fit method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_input_df_contains_na_in_fit(make_df): + df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) transformer = GeometricWidthDiscretiser() with pytest.raises(ValueError): transformer.fit(df_na) -def test_error_if_input_df_contains_na_in_transform(df_vartypes, df_na): - # test case 4: when dataset contains na, transform method +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_error_if_input_df_contains_na_in_transform(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) + df_na = make_df({"Age": [20.0, 21.0, float("nan"), 23.0]}) + transformer = GeometricWidthDiscretiser() - transformer.fit(df_vartypes) + transformer.fit(df) with pytest.raises(ValueError): - transformer.transform(df_na[["Name", "City", "Age", "Marks", "dob"]]) + transformer.transform(df_na) -def test_non_fitted_error(df_vartypes): +@pytest.mark.parametrize("make_df", [pd.DataFrame, pl.DataFrame]) +def test_non_fitted_error(make_df): + df = make_df({"Age": [20.0, 21.0, 19.0, 23.0]}) transformer = GeometricWidthDiscretiser() with pytest.raises(NotFittedError): - transformer.transform(df_vartypes) + transformer.transform(df)