diff --git a/CHANGES.rst b/CHANGES.rst index b3276e3cf..f12bd3c68 100644 --- a/CHANGES.rst +++ b/CHANGES.rst @@ -15,6 +15,10 @@ New Features (of known size), for example ``first, second = data_op``. Each target becomes a DataOp that extracts one of the items. :pr:`2243` by :user:`Elias Strauss `. +-- TabularPipeline now uses the estimator when given a pipeline to determine + the parameters of the TableVectorizer. + :pr:`2152` by :user:`Khaoula Riad and Marine Michaut`. + Changes ------- @@ -165,6 +169,16 @@ New Features :meth:`DataOp.skb.eval`, :meth:`SkrubLearner.predict`, etc., or in :meth:`DataOp.skb.find` or :meth:`SkrubLearner.truncated_after`. :pr:`2062` by :user:`Jérôme Dockès `. +- The :class:`DropSimilar` transformer has been added, for removing columns in a + dataframe that present high correlation with other columns. :pr:`2023` by + :user:`Eloi Massoulié `. +- :class:`ToFloat32` now allows users to specify ``decimal`` and ``thousand`` + separators to parse numerical columns that use formatting different from the default + formatting used in Python, such as ``1'234,5``. + Additionally, negative numbers indicated with parentheses can be converted to the + regular numeric format (``(432)`` becomes ``-432``). :pr:`1772` by :user:`Gabriela + Gómez Jiménez `. + **Misc**: diff --git a/skrub/_docs/modules/default_wrangling/tabular_pipeline.rst b/skrub/_docs/modules/default_wrangling/tabular_pipeline.rst index d2b3067eb..74045e4a2 100644 --- a/skrub/_docs/modules/default_wrangling/tabular_pipeline.rst +++ b/skrub/_docs/modules/default_wrangling/tabular_pipeline.rst @@ -139,7 +139,7 @@ the default table preprocessing: >>> model_pipeline = make_pipeline(PCA(n_components=20), Ridge()) >>> full_pipeline = tabular_pipeline(model_pipeline) >>> [name for name, _ in full_pipeline.steps] -['tablevectorizer', 'simpleimputer', 'squashingscaler', 'pipeline'] +['tablevectorizer', 'simpleimputer', 'squashingscaler', 'pca', 'ridge'] The user-provided estimator pipeline is appended as a single final step. This means that ``tabular_pipeline`` can still decide which preprocessing steps to diff --git a/skrub/_tabular_pipeline.py b/skrub/_tabular_pipeline.py index e0f607272..9e9695de6 100644 --- a/skrub/_tabular_pipeline.py +++ b/skrub/_tabular_pipeline.py @@ -1,7 +1,7 @@ from sklearn import ensemble from sklearn.base import BaseEstimator from sklearn.impute import SimpleImputer -from sklearn.pipeline import make_pipeline +from sklearn.pipeline import Pipeline, make_pipeline from sklearn.preprocessing import OrdinalEncoder from ._datetime_encoder import DatetimeEncoder @@ -48,6 +48,7 @@ def tabular_pipeline(estimator, *, n_jobs=None): Parameters ---------- estimator : {"regressor", "regression", "classifier", "classification"} or sklearn.base.BaseEstimator + or sklearn.pipeline.Pipeline The estimator to use as the final step in the pipeline. Based on the type of estimator, the previous preprocessing steps and their respective parameters are chosen. The possible values are: @@ -59,6 +60,8 @@ def tabular_pipeline(estimator, *, n_jobs=None): :obj:`~sklearn.ensemble.HistGradientBoostingClassifier` is used as the final step; - a scikit-learn estimator: the provided estimator is used as the final step. + - a scikit-learn pipeline : the whole pipeline is kept and usual pre-processing by the TableVectorizer + is added before, depending on the estimator in the last step of the pipeline. n_jobs : int, default=None Number of jobs to run in parallel in the :obj:`TableVectorizer` step. ``None`` @@ -224,7 +227,6 @@ def tabular_pipeline(estimator, *, n_jobs=None): """ # noqa: E501 vectorizer = TableVectorizer(n_jobs=n_jobs) cat_feat_kwargs = {"categorical_features": "from_dtype"} - if isinstance(estimator, str): if estimator in ("classifier", "classification"): return tabular_pipeline( @@ -240,11 +242,21 @@ def tabular_pipeline(estimator, *, n_jobs=None): "If ``estimator`` is a string it should be 'regressor', 'regression'," " 'classifier' or 'classification'." ) + if isinstance(estimator, Pipeline): + # extract all the transforms but separate the last step, + # which is the estimator (only keeping the second item in + # the tuple, the actual transformer/estimator, and not its name) + *user_transformers, (_, estimator) = estimator.steps + else: + # else just create an empty iterable + user_transformers = () + if isinstance(estimator, type) and issubclass(estimator, BaseEstimator): raise TypeError( "tabular_pipeline expects a scikit-learn estimator as its first" - f" argument. Pass an instance of {estimator.__name__} rather than the class" - " itself." + " argument, or a Pipeline containing a scikit-learn estimator. " + f"Pass an instance of {estimator.__name__} rather than" + " the class itself." ) if not isinstance(estimator, BaseEstimator): raise TypeError( @@ -281,6 +293,7 @@ def tabular_pipeline(estimator, *, n_jobs=None): ) else: vectorizer.set_params(datetime=DatetimeEncoder(periodic_encoding="spline")) + steps = [vectorizer] if not is_estimator_from_tabicl: if not get_tags(estimator).input_tags.allow_nan: @@ -288,5 +301,6 @@ def tabular_pipeline(estimator, *, n_jobs=None): if not isinstance(estimator, _TREE_ENSEMBLE_CLASSES): steps.append(SquashingScaler(max_absolute_value=5)) + steps.extend([transformer for _, transformer in user_transformers]) steps.append(estimator) return make_pipeline(*steps) diff --git a/skrub/tests/test_tabular_pipeline.py b/skrub/tests/test_tabular_pipeline.py index 754fc9bef..33ae433bf 100644 --- a/skrub/tests/test_tabular_pipeline.py +++ b/skrub/tests/test_tabular_pipeline.py @@ -1,8 +1,10 @@ import pytest from sklearn import ensemble from sklearn.base import BaseEstimator +from sklearn.decomposition import PCA from sklearn.impute import SimpleImputer -from sklearn.linear_model import Ridge +from sklearn.linear_model import LogisticRegression, Ridge +from sklearn.pipeline import Pipeline from sklearn.preprocessing import OneHotEncoder, OrdinalEncoder from skrub import ( @@ -15,7 +17,13 @@ @pytest.mark.parametrize( - "learner_kind", ["regressor", "regression", "classifier", "classification"] + "learner_kind", + [ + "regressor", + "regression", + "classifier", + "classification", + ], ) def test_default_pipeline(learner_kind): p = tabular_pipeline(learner_kind) @@ -77,6 +85,16 @@ def test_from_dtype(): assert isinstance(p.named_steps["tablevectorizer"].low_cardinality, ToCategorical) +def test_estimator_is_a_pipeline(): + input_learner = LogisticRegression() + sk_pipeline = Pipeline([("pca", PCA()), ("clf", input_learner)]) + tab_pipeline = tabular_pipeline(sk_pipeline) + assert len(tab_pipeline.steps) == 5 + *_, pca, learner = tab_pipeline.named_steps.values() # keep only the last two steps + assert learner is input_learner + assert isinstance(pca, PCA) + + class TabICLClassifier(BaseEstimator): """Dummy class which pretends to be `tabicl.TabICLClassifier`"""