Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions CHANGES.rst
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,10 @@ New Features
(of known size), for example ``first, second = data_op``. Each target becomes
a DataOp that extracts one of the items.
:pr:`2243` by :user:`Elias Strauss <e-strauss>`.
-- TabularPipeline now uses the estimator when given a pipeline to determine
the parameters of the TableVectorizer.
:pr:`2152` by :user:`Khaoula Riad and Marine Michaut`.


Changes
-------
Expand Down Expand Up @@ -165,6 +169,16 @@ New Features
:meth:`DataOp.skb.eval`, :meth:`SkrubLearner.predict`, etc., or in
:meth:`DataOp.skb.find` or :meth:`SkrubLearner.truncated_after`. :pr:`2062` by
:user:`Jérôme Dockès <jeromedockes>`.
- The :class:`DropSimilar` transformer has been added, for removing columns in a
dataframe that present high correlation with other columns. :pr:`2023` by
:user:`Eloi Massoulié <emassoulie>`.
- :class:`ToFloat32` now allows users to specify ``decimal`` and ``thousand``
separators to parse numerical columns that use formatting different from the default
formatting used in Python, such as ``1'234,5``.
Additionally, negative numbers indicated with parentheses can be converted to the
regular numeric format (``(432)`` becomes ``-432``). :pr:`1772` by :user:`Gabriela
Gómez Jiménez <gabrielapgomezji>`.


**Misc**:

Expand Down
2 changes: 1 addition & 1 deletion skrub/_docs/modules/default_wrangling/tabular_pipeline.rst
Original file line number Diff line number Diff line change
Expand Up @@ -139,7 +139,7 @@ the default table preprocessing:
>>> model_pipeline = make_pipeline(PCA(n_components=20), Ridge())
>>> full_pipeline = tabular_pipeline(model_pipeline)
>>> [name for name, _ in full_pipeline.steps]
['tablevectorizer', 'simpleimputer', 'squashingscaler', 'pipeline']
['tablevectorizer', 'simpleimputer', 'squashingscaler', 'pca', 'ridge']

The user-provided estimator pipeline is appended as a single final step. This
means that ``tabular_pipeline`` can still decide which preprocessing steps to
Expand Down
22 changes: 18 additions & 4 deletions skrub/_tabular_pipeline.py
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
from sklearn import ensemble
from sklearn.base import BaseEstimator
from sklearn.impute import SimpleImputer
from sklearn.pipeline import make_pipeline
from sklearn.pipeline import Pipeline, make_pipeline
from sklearn.preprocessing import OrdinalEncoder

from ._datetime_encoder import DatetimeEncoder
Expand Down Expand Up @@ -48,6 +48,7 @@ def tabular_pipeline(estimator, *, n_jobs=None):
Parameters
----------
estimator : {"regressor", "regression", "classifier", "classification"} or sklearn.base.BaseEstimator
or sklearn.pipeline.Pipeline
The estimator to use as the final step in the pipeline. Based on the type of
estimator, the previous preprocessing steps and their respective parameters are
chosen. The possible values are:
Expand All @@ -59,6 +60,8 @@ def tabular_pipeline(estimator, *, n_jobs=None):
:obj:`~sklearn.ensemble.HistGradientBoostingClassifier` is used as the final
step;
- a scikit-learn estimator: the provided estimator is used as the final step.
- a scikit-learn pipeline : the whole pipeline is kept and usual pre-processing by the TableVectorizer
is added before, depending on the estimator in the last step of the pipeline.

n_jobs : int, default=None
Number of jobs to run in parallel in the :obj:`TableVectorizer` step. ``None``
Expand Down Expand Up @@ -224,7 +227,6 @@ def tabular_pipeline(estimator, *, n_jobs=None):
""" # noqa: E501
vectorizer = TableVectorizer(n_jobs=n_jobs)
cat_feat_kwargs = {"categorical_features": "from_dtype"}

if isinstance(estimator, str):
if estimator in ("classifier", "classification"):
return tabular_pipeline(
Expand All @@ -240,11 +242,21 @@ def tabular_pipeline(estimator, *, n_jobs=None):
"If ``estimator`` is a string it should be 'regressor', 'regression',"
" 'classifier' or 'classification'."
)
if isinstance(estimator, Pipeline):
# extract all the transforms but separate the last step,
# which is the estimator (only keeping the second item in
# the tuple, the actual transformer/estimator, and not its name)
*user_transformers, (_, estimator) = estimator.steps
else:
# else just create an empty iterable
user_transformers = ()

if isinstance(estimator, type) and issubclass(estimator, BaseEstimator):
raise TypeError(
"tabular_pipeline expects a scikit-learn estimator as its first"
f" argument. Pass an instance of {estimator.__name__} rather than the class"
" itself."
" argument, or a Pipeline containing a scikit-learn estimator. "
f"Pass an instance of {estimator.__name__} rather than"
" the class itself."
)
if not isinstance(estimator, BaseEstimator):
raise TypeError(
Expand Down Expand Up @@ -281,12 +293,14 @@ def tabular_pipeline(estimator, *, n_jobs=None):
)
else:
vectorizer.set_params(datetime=DatetimeEncoder(periodic_encoding="spline"))

steps = [vectorizer]
if not is_estimator_from_tabicl:
if not get_tags(estimator).input_tags.allow_nan:
steps.append(SimpleImputer(add_indicator=True))
if not isinstance(estimator, _TREE_ENSEMBLE_CLASSES):
steps.append(SquashingScaler(max_absolute_value=5))

steps.extend([transformer for _, transformer in user_transformers])
steps.append(estimator)
return make_pipeline(*steps)
22 changes: 20 additions & 2 deletions skrub/tests/test_tabular_pipeline.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,10 @@
import pytest
from sklearn import ensemble
from sklearn.base import BaseEstimator
from sklearn.decomposition import PCA
from sklearn.impute import SimpleImputer
from sklearn.linear_model import Ridge
from sklearn.linear_model import LogisticRegression, Ridge
from sklearn.pipeline import Pipeline
from sklearn.preprocessing import OneHotEncoder, OrdinalEncoder

from skrub import (
Expand All @@ -15,7 +17,13 @@


@pytest.mark.parametrize(
"learner_kind", ["regressor", "regression", "classifier", "classification"]
"learner_kind",
Comment thread
lisaleemcb marked this conversation as resolved.
[
"regressor",
"regression",
"classifier",
"classification",
],
)
def test_default_pipeline(learner_kind):
p = tabular_pipeline(learner_kind)
Expand Down Expand Up @@ -77,6 +85,16 @@ def test_from_dtype():
assert isinstance(p.named_steps["tablevectorizer"].low_cardinality, ToCategorical)


def test_estimator_is_a_pipeline():
input_learner = LogisticRegression()
sk_pipeline = Pipeline([("pca", PCA()), ("clf", input_learner)])
tab_pipeline = tabular_pipeline(sk_pipeline)
assert len(tab_pipeline.steps) == 5
*_, pca, learner = tab_pipeline.named_steps.values() # keep only the last two steps
assert learner is input_learner
assert isinstance(pca, PCA)


class TabICLClassifier(BaseEstimator):
"""Dummy class which pretends to be `tabicl.TabICLClassifier`"""

Expand Down