diff --git a/docs/core_concepts/outputs_and_inspection.md b/docs/core_concepts/outputs_and_inspection.md index b8b4d80..87c2889 100644 --- a/docs/core_concepts/outputs_and_inspection.md +++ b/docs/core_concepts/outputs_and_inspection.md @@ -94,7 +94,9 @@ Two parameters control the physical layout of the stacked output. `output_format` : One of `"dense"`, `"sparse"`, or `"auto"`. `"auto"` picks sparse when it saves memory (for -example wide one-hot blocks) and dense otherwise. Default `"dense"`. +example wide one-hot blocks) and dense otherwise. The choice is made once at `fit`, from the +training output, and stored in `output_format_`, so every `transform` returns the same container, +even for a single row. Default `"dense"`. `dtype` : The floating-point precision of the output, for example `numpy.float32` to halve memory. diff --git a/docs/tutorials/adaptive_resolution.md b/docs/tutorials/adaptive_resolution.md index 9159e01..6f7870a 100644 --- a/docs/tutorials/adaptive_resolution.md +++ b/docs/tutorials/adaptive_resolution.md @@ -57,15 +57,16 @@ for name, y in [("simple", simple), ("wiggly", wiggly)]: ``` ```text -simple -> selected width 15 -wiggly -> selected width 15 +simple -> selected width 20 +wiggly -> selected width 20 ``` Both widths land inside the `[5, 20]` window without you having to guess a number up front. -The two happen to match here because the underlying CART selector's split count is governed -more by its own tree depth and minimum-samples settings than by how wiggly the signal looks; -with noisier or smaller data, or a narrower window, the two searches can land on different -widths. The bound is what you control directly, the exact count inside it is data-driven. +Here both reach the upper bound: with 3000 samples the CART selector finds more informative +splits than the window admits, so it keeps the most informative ones. Its split count is governed +more by its own tree depth and minimum-samples settings than by how wiggly the signal looks; with +smaller data or a wider window the two searches can land on different widths. The bound is what +you control directly, the exact count inside it is data-driven. ```{note} Fitting a target-aware transformer directly like this, outside a `Pipeline`, normally emits a diff --git a/pretab/compose/config.py b/pretab/compose/config.py index b5c7dc7..a9c5ff2 100644 --- a/pretab/compose/config.py +++ b/pretab/compose/config.py @@ -175,10 +175,11 @@ def imputation_plan(self, *, is_numerical: bool) -> dict: booleans and the imputer ``strategy``. When ``missing_policy`` is ``None`` the explicit ``*_imputation`` / ``add_missing_indicator`` parameters stay authoritative (historical behaviour); otherwise ``missing_policy`` decides. + ``add_indicator`` appends a missing indicator built on the raw input next + to the representation (see :func:`~pretab.compose.factory.create_transformer`). ``add_missing_indicator=True`` with imputation disabled for this column kind routes through the standalone ``MissingStateIndicator`` (via - ``separate_state``) instead of the imputer's own indicator, since - ``SimpleImputer.add_indicator`` only takes effect when the imputer runs. + ``separate_state``) instead, preserving its one-column-per-feature output. """ strategy = ( (self.numerical_imputation or "median") diff --git a/pretab/compose/factory.py b/pretab/compose/factory.py index 8ced4d6..9c14479 100644 --- a/pretab/compose/factory.py +++ b/pretab/compose/factory.py @@ -10,8 +10,9 @@ import warnings +import numpy as np from sklearn.compose import ColumnTransformer -from sklearn.impute import SimpleImputer +from sklearn.impute import MissingIndicator, SimpleImputer from sklearn.pipeline import FeatureUnion, Pipeline from sklearn.preprocessing import MinMaxScaler, StandardScaler @@ -49,6 +50,20 @@ _KNOT_SPLINE_METHODS = _BMI_SPLINE_METHODS | frozenset({"cubicspline", "naturalspline"}) +class _PositiveMinMaxScaler(MinMaxScaler): + """MinMaxScaler whose output is floored at ``feature_range[0]``. + + Rescales into ``(1e-3, 1)`` ahead of Box-Cox, which requires strictly + positive input. Fitted on the training data, a plain MinMaxScaler maps an + unseen value below the training minimum to ``<= 0``; flooring only the lower + end makes such values transform like the training minimum while every other + output (including values above the training maximum) is unchanged. + """ + + def transform(self, X): + return np.maximum(super().transform(X), self.feature_range[0]) + + def _filter_kwargs(allowed, kwargs): """Keep only the ``allowed`` keyword arguments that are present in ``kwargs``.""" return {key: kwargs[key] for key in allowed if key in kwargs} @@ -154,7 +169,7 @@ def get_numerical_transformer_steps( placement = _placement_kwargs(spec, kwargs) if method == "box-cox": - steps.append(("scale_positive", MinMaxScaler(feature_range=(1e-3, 1)))) + steps.append(("scale_positive", _PositiveMinMaxScaler(feature_range=(1e-3, 1)))) steps.append(("boxcox", cls(method="box-cox", **filtered))) elif method == "yeo-johnson": steps.append(("yeojohnson", cls(method="yeo-johnson", **filtered))) @@ -263,7 +278,6 @@ def create_transformer(method: str, *, is_numerical: bool, config: PreprocessorC target_aware=config.target_aware, add_imputer=plan["add_imputer"], imputer_strategy=plan["strategy"], - add_missing_indicator=plan["add_indicator"], output_dim=config.output_dim, adaptive=config.adaptive, min_output_dim=config.min_output_dim if config.adaptive else None, @@ -293,7 +307,6 @@ def create_transformer(method: str, *, is_numerical: bool, config: PreprocessorC method, add_imputer=plan["add_imputer"], imputer_strategy=plan["strategy"], - add_missing_indicator=plan["add_indicator"], **constructor_kwargs, ) @@ -302,9 +315,44 @@ def create_transformer(method: str, *, is_numerical: bool, config: PreprocessorC # Emit a dedicated ``__missing`` column (built on the raw input) alongside # the imputed representation, so the indicator never enters the basis. return FeatureUnion([("representation", pipeline), ("missing", MissingStateIndicator())]) + if plan["add_indicator"]: + # The imputer's missing indicator, built on the raw input next to the + # representation instead of being fed through the scaler and basis as a + # second feature. It keeps SimpleImputer(add_indicator=True)'s semantics + # (a column only for features with missing values at fit) and names. + indicator = MissingIndicator(error_on_new=False) + return FeatureUnion( + [("representation", pipeline), ("missing", indicator)], + verbose_feature_names_out=False, + ) return pipeline +def _step_names(blocks) -> list[str]: + """Return one valid, unique ColumnTransformer step name per ``(kind, feature)`` block. + + A step is named ``f"{kind}_{feature}"`` whenever scikit-learn accepts that + name. A label that would put scikit-learn's ``__`` parameter separator into it + (one starting with ``_`` or containing ``__``) gets a positional + ``f"{kind}_col{i}"`` name instead, made unique against every other step. The + public block / feature names never use these internal names: they are derived + from the step's kind and its column label (see :mod:`pretab.compose.inspection`). + """ + natural = [f"{kind}_{feature}" for kind, feature in blocks] + taken = {name for name in natural if "__" not in name} + names = [] + for position, name in enumerate(natural): + if "__" in name: + kind = blocks[position][0] + name, suffix = f"{kind}_col{position}", 0 + while name in taken: + suffix += 1 + name = f"{kind}_col{position}_{suffix}" + taken.add(name) + names.append(name) + return names + + def build_column_transformer( config: PreprocessorConfig, numerical_features, @@ -314,19 +362,22 @@ def build_column_transformer( ) -> ColumnTransformer: """Assemble the per-column pipelines into the final ColumnTransformer. - Numerical features are prefixed ``num_`` and categorical features ``cat_`` to - match the transformer names the Preprocessor exposes; untransformed columns - pass through via ``remainder="passthrough"``. + Numerical steps are prefixed ``num_`` and categorical steps ``cat_`` to match + the block names the Preprocessor exposes (see :func:`_step_names` for labels + that cannot be embedded in a step name); untransformed columns pass through via + ``remainder="passthrough"``. Each step selects its column by the label's + string form: scikit-learn reads an integer selector as a *position*, so the + Preprocessor fits and transforms the ColumnTransformer on a frame whose labels + are strings (see :func:`~pretab.compose.feature_detection.with_string_labels`). """ + blocks = [("num", feature) for feature in numerical_features] + blocks += [("cat", feature) for feature in categorical_features] transformers = [] - for feature in numerical_features: - method = config.method_for(feature, is_numerical=True) - pipeline = create_transformer(method, is_numerical=True, config=config) - transformers.append((f"num_{feature}", pipeline, [feature])) - for feature in categorical_features: - method = config.method_for(feature, is_numerical=False) - pipeline = create_transformer(method, is_numerical=False, config=config) - transformers.append((f"cat_{feature}", pipeline, [feature])) + for (kind, feature), name in zip(blocks, _step_names(blocks), strict=True): + is_numerical = kind == "num" + method = config.method_for(feature, is_numerical=is_numerical) + pipeline = create_transformer(method, is_numerical=is_numerical, config=config) + transformers.append((name, pipeline, [str(feature)])) return ColumnTransformer( transformers=transformers, remainder="passthrough", diff --git a/pretab/compose/feature_detection.py b/pretab/compose/feature_detection.py index ebdb1cc..5f1cec3 100644 --- a/pretab/compose/feature_detection.py +++ b/pretab/compose/feature_detection.py @@ -5,12 +5,14 @@ sequence of delegations rather than inlining the classification heuristic. """ +from collections import Counter + import numpy as np import pandas as pd from ..exceptions import PretabDataError, invalid_param_error -__all__ = ["detect_column_types", "to_dataframe"] +__all__ = ["bool_columns_as_object", "detect_column_types", "to_dataframe", "with_string_labels"] def to_dataframe(X, *, copy: bool = False) -> pd.DataFrame: @@ -41,6 +43,58 @@ def to_dataframe(X, *, copy: bool = False) -> pd.DataFrame: return X +def with_string_labels(X: pd.DataFrame) -> pd.DataFrame: + """Return ``X`` with every column label converted to its ``str`` form. + + scikit-learn's :class:`~sklearn.compose.ColumnTransformer` treats an integer + column selector as a *position*, so routing a column by an integer label (as + in ``pd.DataFrame(array)`` or ``read_csv(header=None)`` after reordering or + dropping a column) would select the wrong column. The Preprocessor therefore + fits and transforms its ColumnTransformer on string labels. ``X`` itself is + returned when every label is already a string; otherwise a shallow copy with + relabelled columns, so the caller's frame is never modified. + + Raises + ------ + PretabDataError + If two labels share a string form (e.g. ``1`` and ``"1"``), which would + make the columns indistinguishable. + """ + if all(isinstance(label, str) for label in X.columns): + return X + labels = [str(label) for label in X.columns] + collisions = sorted(label for label, count in Counter(labels).items() if count > 1) + if collisions: + raise PretabDataError( + f"Column labels must stay unique when converted to strings; {collisions} would collide.\n" + "Fix: rename the columns so their string forms are unique." + ) + relabelled = X.copy(deep=False) + relabelled.columns = pd.Index(labels) + return relabelled + + +def bool_columns_as_object(X: pd.DataFrame) -> pd.DataFrame: + """Return ``X`` with its boolean columns cast to ``object``. + + Boolean columns are categorical (see :func:`detect_column_types`), but the + categorical pipeline starts with a :class:`~sklearn.impute.SimpleImputer`, + which rejects the ``bool`` dtype. As ``object`` columns of ``True`` / + ``False`` they are encoded like any other binary categorical column; a + missing value of pandas' nullable ``boolean`` dtype becomes ``NaN``, the + missing marker the imputer recognizes. ``X`` itself is returned when it has + no boolean column, so the caller's frame is never modified. + """ + bool_columns = [label for label, dtype in X.dtypes.items() if dtype.kind == "b"] + if not bool_columns: + return X + cast = X.copy(deep=False) + for label in bool_columns: + column = X[label].astype(object) + cast[label] = column.where(column.notna(), np.nan) + return cast + + def detect_column_types(X, *, cat_cutoff, treat_all_integers_as_numerical, estimator_name="Preprocessor"): """Classify each column of ``X`` as numerical or categorical. diff --git a/pretab/compose/inspection.py b/pretab/compose/inspection.py index 7f3b9b3..a5d8efa 100644 --- a/pretab/compose/inspection.py +++ b/pretab/compose/inspection.py @@ -7,8 +7,11 @@ :func:`build_transformer_summary` renders that metadata as an aligned table. """ +import copy +from typing import Any + import numpy as np -from sklearn.pipeline import FeatureUnion +from sklearn.pipeline import FeatureUnion, Pipeline from ..core.logging import get_logger from ..core.representation import FeatureLineage @@ -16,23 +19,48 @@ logger = get_logger(__name__) __all__ = [ + "block_name", "build_feature_info", "build_feature_lineage", "build_transformer_summary", "clean_feature_names", + "feature_names_out", "get_output_slices", ] +def _block(step_name, columns): + """Return ``(kind, feature)`` for a per-column ``num_*`` / ``cat_*`` step, else ``None``.""" + if step_name == "remainder" or len(columns) != 1: + return None + kind, sep, _ = str(step_name).partition("_") + if not sep or kind not in ("num", "cat"): + return None + return kind, str(columns[0]) + + +def block_name(step_name, columns): + """Return the public block name (``"num_