diff --git a/docs/representations/spline_expansions.md b/docs/representations/spline_expansions.md index d6b28d7..9d813a9 100644 --- a/docs/representations/spline_expansions.md +++ b/docs/representations/spline_expansions.md @@ -23,8 +23,8 @@ knot positions by `placement_strategy` (see For every univariate spline in this section, `output_dim` is the number of output columns **per input feature**, not the total. A `(n_samples, 3)` input produces `(n_samples, 3 * output_dim)` output (plus one extra column per feature if -`include_bias=True`). Feature names are suffixed per input, for example `x0_bs0, x0_bs1, ...` -for a B-spline on column `x0`. +`include_bias=True`). Feature names are suffixed per input, for example `age_bs0, age_bs1, ...` +for a B-spline fitted on a DataFrame column `age` (`x0_bs0, x0_bs1, ...` for array input). ``` ## B-spline diff --git a/pretab/compose/inspection.py b/pretab/compose/inspection.py index a5d8efa..8d53611 100644 --- a/pretab/compose/inspection.py +++ b/pretab/compose/inspection.py @@ -11,21 +11,25 @@ from typing import Any import numpy as np +import pandas as pd from sklearn.pipeline import FeatureUnion, Pipeline from ..core.logging import get_logger from ..core.representation import FeatureLineage +from .registry import TRANSFORMER_REGISTRY logger = get_logger(__name__) __all__ = [ "block_name", + "block_uses_target", "build_feature_info", "build_feature_lineage", "build_transformer_summary", "clean_feature_names", "feature_names_out", "get_output_slices", + "representation_leaf", ] @@ -177,6 +181,32 @@ def _separate_state_branches(transformer): return branches["representation"], branches["missing"] +def representation_leaf(transformer): + """Return the last step of a fitted per-column block's representation. + + For a separate-state / missing-indicator union this is the last step of its + ``"representation"`` pipeline (the raw missing indicator beside it is not + part of the representation); for a pipeline it is the last step, and any + other block is returned unchanged. + """ + separate_state = _separate_state_branches(transformer) + representation = transformer if separate_state is None else separate_state[0] + return representation.steps[-1][1] if hasattr(representation, "steps") else representation + + +def _probe_input(step, value): + """Return a one-row probe input for measuring a fitted step's output width. + + A step fitted directly on the ColumnTransformer's DataFrame column records + ``feature_names_in_`` and, like scikit-learn, warns on input without feature + names, so its probe is a DataFrame carrying the fitted names. + """ + names = getattr(step, "feature_names_in_", None) + if names is None: + return np.full((1, 1), value) + return pd.DataFrame(np.full((1, len(names)), value), columns=names) + + def build_feature_info(column_transformer, *, embeddings, embedding_dimensions): """Collect per-feature metadata (preprocessing, dimension, categories). @@ -230,7 +260,7 @@ def build_feature_info(column_transformer, *, embeddings, embedding_dimensions): ): last_step = representation_pipeline.steps[-1][1] if hasattr(last_step, "transform"): - dummy_input = np.zeros((1, 1)) + 1e-05 + dummy_input = _probe_input(last_step, 1e-05) try: transformed_feature = last_step.transform(dummy_input) dimension = transformed_feature.shape[1] @@ -275,7 +305,7 @@ def build_feature_info(column_transformer, *, embeddings, embedding_dimensions): else: last_step = representation_pipeline.steps[-1][1] if hasattr(last_step, "transform"): - dummy_input = np.zeros((1, 1)) + dummy_input = _probe_input(last_step, 0.0) try: transformed_feature = last_step.transform(dummy_input) dimension = transformed_feature.shape[1] @@ -348,8 +378,9 @@ def _resolve_block_representation(pipeline, columns): """Return ``(family, component, uses_target, is_interaction)`` for a block. The representation-bearing step is the last pipeline step exposing a - ``get_representation_spec`` (a PreTab transformer) or a known scikit-learn - step name; helper steps such as imputers and float casts are skipped. + ``get_representation_spec`` (a PreTab transformer), a known scikit-learn + step name, or a registered method that consumes the target; helper steps + such as imputers and float casts are skipped. """ steps = pipeline.steps if hasattr(pipeline, "steps") else [("_", pipeline)] for step_name, transformer in reversed(steps): @@ -359,9 +390,28 @@ def _resolve_block_representation(pipeline, columns): if step_name in _STEP_FAMILY: family, component = _STEP_FAMILY[step_name] return family, component, False, False + registered = TRANSFORMER_REGISTRY.get(step_name) + if registered is not None and registered.target_usage != "forbidden": + # A registered class without a RepresentationSpec (e.g. scikit-learn's + # TargetEncoder): its declared supervision says whether it used y, so + # cross-fitting refits it per fold instead of reusing the all-data fit. + uses_target = registered.target_usage == "required" or bool(getattr(transformer, "target_aware", False)) + component = "category" if registered.is_categorical else "basis" + return step_name, component, uses_target, registered.is_multivariate return "passthrough", "raw", False, False +def block_uses_target(transformer, columns) -> bool: + """Whether a fitted per-column block's representation consumed the target ``y``. + + For a separate-state / missing-indicator union the representation branch + decides; helper steps such as imputers and scalers never use the target. + """ + separate_state = _separate_state_branches(transformer) + representation = separate_state[0] if separate_state is not None else transformer + return bool(_resolve_block_representation(representation, columns)[2]) + + def _passthrough_source(columns, offset, feature_names_in): """Resolve the source feature name for a passthrough / remainder column.""" column = columns[offset] if offset < len(columns) else columns[-1] diff --git a/pretab/compose/search.py b/pretab/compose/search.py index 7ce1d63..f01a140 100644 --- a/pretab/compose/search.py +++ b/pretab/compose/search.py @@ -47,7 +47,9 @@ class RepresentationSearchCV(BaseEstimator): Scoring passed to :func:`sklearn.metrics.check_scoring`; ``None`` uses the estimator's ``score`` method. preprocessor_params : dict or None, default=None - Extra keyword arguments forwarded to every :class:`Preprocessor`. + Extra keyword arguments forwarded to every :class:`Preprocessor`. When + ``estimator`` is a classifier, ``task`` defaults to ``"classification"`` so + target-aware placement treats ``y`` as class labels. random_state : int or None, default=None Seed forwarded to each :class:`Preprocessor`. @@ -74,9 +76,15 @@ def __init__(self, estimator, methods, *, cv=5, scoring=None, preprocessor_param self.random_state = random_state def _make_preprocessor(self, method): - """Build a Preprocessor for ``method`` with the shared parameters.""" + """Build a Preprocessor for ``method`` with the shared parameters. + + Target-aware placement follows the downstream estimator: a classifier makes + it ``task="classification"`` unless ``preprocessor_params`` sets ``task``. + """ params = dict(self.preprocessor_params or {}) params.setdefault("random_state", self.random_state) + if is_classifier(self.estimator): + params.setdefault("task", "classification") return Preprocessor(numerical_method=method, **params) def fit(self, X, y=None): diff --git a/pretab/compose/serialize.py b/pretab/compose/serialize.py index 025fbef..2b85f5a 100644 --- a/pretab/compose/serialize.py +++ b/pretab/compose/serialize.py @@ -16,6 +16,7 @@ import dataclasses import importlib +import math from collections import UserList from typing import Any, cast @@ -27,8 +28,9 @@ from ..core.parameters import UNSET from ..core.policy import RepresentationPolicy from ..core.representation import FeatureLineage, RepresentationSpec -from ..exceptions import PretabError, PretabSerializationError +from ..exceptions import PretabSerializationError from ..placement.base import PlacementResult +from .inspection import representation_leaf from .registry import TransformerSpec SCHEMA_VERSION = 1 @@ -88,14 +90,23 @@ def _resolve(dotted: str): # --- encoding ------------------------------------------------------------ +def _encode_float(value): + """Return ``value``, tagging NaN / +-inf, which strict JSON cannot represent.""" + if math.isfinite(value): + return value + return {"__float__": repr(float(value))} + + def _encode(obj): if isinstance(obj, np.bool_): return bool(obj) if isinstance(obj, np.integer): return int(obj) if isinstance(obj, np.floating): - return float(obj) - if obj is None or isinstance(obj, (bool, int, float, str)): + return _encode_float(float(obj)) + if isinstance(obj, float): + return _encode_float(obj) + if obj is None or isinstance(obj, (bool, int, str)): return obj if obj is UNSET: return {"__unset__": True} @@ -105,6 +116,8 @@ def _encode(obj): # tolist() keeps the elements of an object array as they are (e.g. # numpy scalars such as np.bool_), so encode each one. data = _map_elements(data, obj.ndim, _encode) + elif obj.dtype.kind == "f" and not np.isfinite(obj).all(): + data = _map_elements(data, obj.ndim, _encode_float) return {"__ndarray__": {"dtype": obj.dtype.str, "shape": list(obj.shape), "data": data}} if isinstance(obj, np.dtype): return {"__npdtype__": obj.str} @@ -151,7 +164,9 @@ def _json_safe(obj): Keeps JSON-native values verbatim and falls back to the tagged :func:`_encode` form only for exotic values. The result is informational and never decoded. """ - if obj is None or isinstance(obj, (bool, int, float, str)): + if isinstance(obj, float): + return _encode_float(obj) + if obj is None or isinstance(obj, (bool, int, str)): return obj if isinstance(obj, dict) and all(isinstance(k, str) for k in obj): return {k: _json_safe(v) for k, v in obj.items()} @@ -173,7 +188,10 @@ def _decode_ndarray(payload: dict) -> np.ndarray: for index, element in enumerate(elements): arr[index] = element return arr.reshape(shape) - arr = np.array(payload["data"], dtype=dtype) + data = payload["data"] + if dtype.kind == "f": + data = _map_elements(data, len(payload["shape"]), _decode) + arr = np.array(data, dtype=dtype) return arr.reshape(payload["shape"]) @@ -187,6 +205,8 @@ def _decode(obj): return _decode_ndarray(obj["__ndarray__"]) if "__unset__" in obj: return UNSET + if "__float__" in obj: + return float(obj["__float__"]) if "__npdtype__" in obj: return np.dtype(obj["__npdtype__"]) if "__type__" in obj: @@ -262,7 +282,17 @@ def _library_versions() -> dict: def _representation_summary(preprocessor) -> list: - """Best-effort declarative per-representation summary (family/columns/locations).""" + """Declarative per-representation summary (family/columns/locations). + + One entry per block whose representation ends in a PreTab transformer (for a + missing-indicator / separate-state block, its representation branch; see + :func:`~pretab.compose.inspection.representation_leaf`), built from its + :class:`RepresentationSpec` with the block's column names as input features + (as the feature lineage does). Blocks whose representation ends in a step + without ``get_representation_spec`` (scikit-learn scalers and encoders, + embeddings) have no entry. The steps are fitted, so a spec that fails to + build is a bug: its error propagates instead of silently dropping the entry. + """ summary: list = [] column_transformer = getattr(preprocessor, "column_transformer_", None) if column_transformer is None: @@ -270,15 +300,17 @@ def _representation_summary(preprocessor) -> list: for name, transformer, columns in column_transformer.transformers_: if name == "remainder": continue - leaf = transformer.steps[-1][1] if hasattr(transformer, "steps") else transformer + leaf = representation_leaf(transformer) spec_fn = getattr(leaf, "get_representation_spec", None) if spec_fn is None: continue - try: - entry = spec_fn().to_dict() - except (PretabError, ValueError, AttributeError, TypeError, KeyError): - continue - entry["columns"] = [str(col) for col in columns] + block_columns = [str(col) for col in columns] + # A leaf with more inputs than its block has columns (one fitted behind + # the imputer's built-in missing indicator, as before the #62 fix) names + # its own inputs. + fits_block = getattr(leaf, "n_features_in_", len(block_columns)) == len(block_columns) + entry = spec_fn(input_features=block_columns if fits_block else None).to_dict() + entry["columns"] = block_columns summary.append(entry) return summary diff --git a/pretab/core/base.py b/pretab/core/base.py index 284a0da..d12f3cd 100644 --- a/pretab/core/base.py +++ b/pretab/core/base.py @@ -10,12 +10,11 @@ from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted -from ..exceptions import invalid_param_error from .adaptive import AdaptiveResolutionMixin from .parameters import AliasResolverMixin from .policy import RepresentationPolicy, apply_constant_policy from .representation import RepresentationSpecMixin -from .validation import validate_2d_allow_nan +from .validation import resolve_input_features, validate_2d_allow_nan __all__ = ["BasePreTabTransformer"] @@ -58,19 +57,24 @@ class BasePreTabTransformer( def _resolved_policy(self) -> RepresentationPolicy: """Return the effective edge-case policy for this instance. - An explicit ``policy`` constructor argument, on the transformers that expose - one, always wins verbatim over the class-level defaults below. Otherwise, the - shared default policy is narrowed by this family's ``_constant_policy`` / - ``_out_of_range_policy`` class attributes, which record each family's - historical, non-configurable default behavior. + The shared default policy is narrowed by this family's ``_constant_policy`` + / ``_out_of_range_policy`` class attributes, which record each family's + historical default behavior. An explicit ``policy`` constructor argument, on + the transformers that expose one, then applies on top: a mapping overrides + only the axes it names (so ``{"constant": "error"}`` keeps the family's + out-of-range handling), and a :class:`RepresentationPolicy` instance is used + verbatim. """ - instance_policy = getattr(self, "policy", None) - if instance_policy is not None: - return RepresentationPolicy.resolve(instance_policy) - return self._policy.merge( + family_policy = self._policy.merge( constant=self._constant_policy, out_of_range=self._out_of_range_policy, ) + instance_policy = getattr(self, "policy", None) + if instance_policy is None: + return family_policy + if isinstance(instance_policy, dict): + return family_policy.merge(**instance_policy) + return RepresentationPolicy.resolve(instance_policy) def _validate(self, X, *, reset: bool): """Validate ``X`` through the shared NaN-aware validator.""" @@ -90,15 +94,7 @@ def _output_sizes(self) -> list[int]: def get_feature_names_out(self, input_features=None): """Return output feature names of the form ``{feature}_{suffix}{j}``.""" check_is_fitted(self, "n_features_in_") - if input_features is None: - input_features = [f"x{i}" for i in range(self.n_features_in_)] - elif len(input_features) != self.n_features_in_: - raise invalid_param_error( - type(self).__name__, - "get_feature_names_out.input_features", - len(input_features), - f"must have exactly {self.n_features_in_} entries (one per input feature)", - ) + input_features = resolve_input_features(self, input_features) suffix = self._feature_suffix() names = [] for feature, n_cols in zip(input_features, self._output_sizes(), strict=False): diff --git a/pretab/core/knots.py b/pretab/core/knots.py index 3bf81c2..0690d27 100644 --- a/pretab/core/knots.py +++ b/pretab/core/knots.py @@ -19,6 +19,7 @@ __all__ = [ "basis_to_knots", "bspline_basis", + "extrapolate_bspline_rows", "generate_internal_knots", "quantile_knots", "select_knots", @@ -58,6 +59,23 @@ def bspline_basis(x: np.ndarray, knots: np.ndarray, degree: int, i: int, last: i return term1 + term2 +def extrapolate_bspline_rows(basis: np.ndarray, x: np.ndarray, knots: np.ndarray, degree: int) -> np.ndarray: + """Fill the rows of ``basis`` whose ``x`` lies outside the knot span by extrapolation. + + :func:`bspline_basis` is zero outside ``[knots[0], knots[-1]]``. When a policy lets + out-of-range values through (``"extrapolate"`` / ``"warn"``), those rows are + evaluated instead by extending the boundary polynomial pieces, which keeps the + basis a partition of unity there. Rows inside the span are left untouched. + """ + outside = (x < knots[0]) | (x > knots[-1]) + if outside.any(): + from scipy.interpolate import BSpline + + n_basis = len(knots) - degree - 1 + basis[outside] = BSpline(knots, np.eye(n_basis), degree, extrapolate=True)(x[outside]) + return basis + + def basis_to_knots(n_basis: int, degree: int) -> int: """Number of internal knots implied by ``n_basis`` basis functions of ``degree``.""" return max(0, n_basis - degree - 1) diff --git a/pretab/core/policy.py b/pretab/core/policy.py index 10a27ec..9c49484 100644 --- a/pretab/core/policy.py +++ b/pretab/core/policy.py @@ -10,6 +10,8 @@ The defaults reproduce each family's historical behaviour (B/M/I-spline, P-spline, and tensor-product clip out-of-range inputs; natural-cubic and cubic-regression extrapolate; constant columns pass through everywhere), so leaving ``policy`` unset changes nothing. +A transformer's ``policy`` given as a mapping overrides only the axes it names on top of +those family defaults; a :class:`RepresentationPolicy` instance is used verbatim. Transformers may narrow specific axes through class-level override attributes without exposing a new constructor parameter (see :class:`~pretab.core.base.BasePreTabTransformer`). diff --git a/pretab/core/representation.py b/pretab/core/representation.py index 0c2676b..bf1bf8e 100644 --- a/pretab/core/representation.py +++ b/pretab/core/representation.py @@ -15,6 +15,8 @@ import numpy as np from sklearn.utils.validation import check_is_fitted +from .validation import resolve_input_features + __all__ = [ "FeatureLineage", "RepresentationSpec", @@ -297,8 +299,11 @@ def get_representation_spec(self, input_features=None) -> RepresentationSpec: Parameters ---------- input_features : list of str or None - Names of the input features. When ``None``, names of the form - ``x0, x1, ...`` are generated. + Names of the input features, resolved as in ``get_feature_names_out``: + explicit names need one entry per input feature and, after a fit on + named (DataFrame) columns, must equal ``feature_names_in_``. When + ``None``, ``feature_names_in_`` is used if it was recorded at fit, + else names of the form ``x0, x1, ...`` are generated. Returns ------- @@ -306,10 +311,7 @@ def get_representation_spec(self, input_features=None) -> RepresentationSpec: The representation metadata describing the produced columns. """ check_is_fitted(self, "n_features_in_") - if input_features is None: - inputs = tuple(f"x{i}" for i in range(self.n_features_in_)) - else: - inputs = tuple(str(feature) for feature in input_features) + inputs = tuple(resolve_input_features(self, input_features)) output_features = tuple(str(name) for name in self.get_feature_names_out(list(inputs))) location_kind, locations = self._representation_locations() periodic, period = self._representation_periodic() diff --git a/pretab/core/supervised.py b/pretab/core/supervised.py index a6f0531..b221b10 100644 --- a/pretab/core/supervised.py +++ b/pretab/core/supervised.py @@ -19,6 +19,7 @@ from typing import cast import numpy as np +from scipy import sparse as sp from sklearn.base import BaseEstimator, TransformerMixin, clone from sklearn.model_selection import KFold, StratifiedKFold from sklearn.utils.validation import check_is_fitted @@ -120,7 +121,11 @@ class CrossFittedTransformer(RepresentationSpecMixin, TransformerMixin, BaseEsti A supervised (target-aware) PreTab transformer to cross-fit, or an estimator such as a :class:`~pretab.Preprocessor` or ``Pipeline`` built from them. A pandas DataFrame ``X`` is passed to it unchanged (each fold is a row - subset), so column names and dtypes are preserved. + subset), so column names and dtypes are preserved. For a + :class:`~pretab.Preprocessor`, only its target-aware blocks are refit per + fold: column types, category codes and the blocks that do not use ``y`` + come from the all-data fit, so the out-of-fold features use exactly the + encoding of :meth:`transform`. n_folds : int, default=5 Number of cross-fitting folds. Must be at least 2. task : {"regression", "classification"}, default="regression" @@ -187,6 +192,21 @@ def transform(self, X): check_is_fitted(self, "estimator_") return self.estimator_.transform(_as_2d(X)) + def _fit_fold(self, X, y): + """Return the model that encodes one held-out fold, fit on the other rows. + + A wrapped estimator that implements ``_cross_fit_fold`` (the + :class:`~pretab.Preprocessor`) derives the fold model from ``estimator_`` + and refits only its target-aware parts, so the out-of-fold features share + the encoding ``transform`` uses. Any other estimator is cloned and refit. + """ + cross_fit_fold = getattr(self.estimator_, "_cross_fit_fold", None) + if cross_fit_fold is not None: + return cast(TransformerLike, cross_fit_fold(X, y)) + fold = cast(TransformerLike, clone(self.transformer)) + fold.fit(X, y) + return fold + def fit_transform(self, X, y=None): """Fit and return leakage-free out-of-fold features for the training data.""" X_arr, y_arr = self._fit_full(X, y) @@ -196,14 +216,21 @@ def fit_transform(self, X, y=None): token = _cross_fit_active.set(True) try: for train_idx, test_idx in splitter.split(X_arr, y_arr): - fold = cast(TransformerLike, clone(self.transformer)) - fold.fit(_take_rows(X_arr, train_idx), y_arr[train_idx]) - fold_out = np.asarray(fold.transform(_take_rows(X_arr, test_idx))) + fold = self._fit_fold(_take_rows(X_arr, train_idx), y_arr[train_idx]) + fold_out = fold.transform(_take_rows(X_arr, test_idx)) + if isinstance(fold_out, dict): + raise IncompatibleParamsError( + "Cross-fitting stacks the out-of-fold features into one array; the " + "wrapped transformer returned a dict of blocks. Use output_structure=" + "'matrix' on a wrapped Preprocessor." + ) + fold_out = fold_out.toarray() if sp.issparse(fold_out) else np.asarray(fold_out) if fold_out.shape[1] != width: raise IncompatibleParamsError( "Cross-fitting requires a fixed output width across folds; expected " - f"{width}, got {fold_out.shape[1]}. Disable adaptive sizing on the " - "wrapped transformer." + f"{width}, got {fold_out.shape[1]}. The wrapped transformer's width " + "depends on the training rows (e.g. adaptive sizing); fix it, for " + "example by disabling adaptive sizing." ) out[test_idx] = fold_out finally: @@ -222,6 +249,10 @@ def get_representation_spec(self, input_features=None): if spec_fn is not None: base = spec_fn(input_features) return replace(base, uses_target=True, cross_fitted=True, n_folds=int(self.n_folds)) + if input_features is None: + # Name the inputs as the wrapped estimator does, so a DataFrame fit + # passes its column names (not x0, x1, ...) to get_feature_names_out. + input_features = getattr(self.estimator_, "feature_names_in_", None) return super().get_representation_spec(input_features) def _representation_cross_fitting(self): diff --git a/pretab/core/validation.py b/pretab/core/validation.py index fbe7ea7..8a271e7 100644 --- a/pretab/core/validation.py +++ b/pretab/core/validation.py @@ -7,14 +7,34 @@ """ import warnings -from typing import Literal +from typing import Literal, cast import numpy as np -from sklearn.utils.validation import check_array +from sklearn.utils.validation import _check_feature_names, _check_feature_names_in, check_array -from ..exceptions import DataWarning, PretabDataError +from ..exceptions import DataWarning, PretabDataError, invalid_param_error -__all__ = ["validate_2d_allow_nan"] +__all__ = ["resolve_input_features", "validate_2d_allow_nan"] + + +def resolve_input_features(estimator, input_features) -> list: + """Return the input feature names that a transformer's output names are built from. + + Follows scikit-learn's ``get_feature_names_out`` contract: explicit + ``input_features`` need one entry per input feature and, when the estimator was + fitted on named (DataFrame) columns, must equal ``feature_names_in_``; ``None`` + falls back to ``feature_names_in_``, else to ``x0, x1, ...``. + """ + n_features_in_ = getattr(estimator, "n_features_in_", None) + if input_features is not None and n_features_in_ is not None and len(input_features) != n_features_in_: + raise invalid_param_error( + type(estimator).__name__, + "get_feature_names_out.input_features", + len(input_features), + f"must have exactly {n_features_in_} entries (one per input feature)", + ) + names = cast(np.ndarray, _check_feature_names_in(estimator, input_features)) + return [str(name) for name in names] def validate_2d_allow_nan(X, *, allow_nan: bool = True, reset: bool, estimator): @@ -28,9 +48,11 @@ def validate_2d_allow_nan(X, *, allow_nan: bool = True, reset: bool, estimator): When True, missing values are preserved so a later imputer can handle them; when False, NaN/inf values raise as usual. reset : bool - When True (during ``fit``) record ``estimator.n_features_in_``; when - False (during ``transform``) verify the feature count matches the value - seen at ``fit`` and raise otherwise. + When True (during ``fit``) record ``estimator.n_features_in_`` and, for a + DataFrame, ``estimator.feature_names_in_``; when False (during + ``transform``) verify the feature count matches the value seen at ``fit`` + and raise otherwise, and check the column names as scikit-learn does + (renamed or reordered columns raise). estimator : object The calling transformer; used for the ``n_features_in_`` side effect and the transform-time feature-count check. @@ -40,6 +62,7 @@ def validate_2d_allow_nan(X, *, allow_nan: bool = True, reset: bool, estimator): X : ndarray of shape (n_samples, n_features) The validated float array. """ + _check_feature_names(estimator, X, reset=reset) input_shape = getattr(X, "shape", None) original_dim = input_shape[1] if input_shape is not None and len(input_shape) == 2 else None ensure_all_finite: Literal["allow-nan"] | bool = "allow-nan" if allow_nan else True diff --git a/pretab/encoding/numerical/binning.py b/pretab/encoding/numerical/binning.py index 4b1fcbf..ff48c1a 100644 --- a/pretab/encoding/numerical/binning.py +++ b/pretab/encoding/numerical/binning.py @@ -2,10 +2,11 @@ import numpy as np from sklearn.base import BaseEstimator, TransformerMixin -from sklearn.utils.validation import check_is_fitted +from sklearn.utils.validation import _check_feature_names, check_is_fitted from ...core.parameters import UNSET, AliasResolverMixin from ...core.representation import RepresentationSpecMixin +from ...core.validation import resolve_input_features from ...exceptions import InsufficientSamplesError, InvalidParamError, PretabDataError _VALID_ENCODINGS = ("ordinal", "onehot", "soft") @@ -89,7 +90,8 @@ def __init__(self, output_dim=UNSET, encode="ordinal", placement_strategy="unifo self.placement_strategy = placement_strategy def _check_array(self, X, *, reset): - """Validate ``X`` is a 2D numeric array and (re)set the feature count.""" + """Validate ``X`` is a 2D numeric array and (re)set the feature count and names.""" + _check_feature_names(self, X, reset=reset) X = np.asarray(X) if X.ndim != 2: raise PretabDataError("Input must be a 2D array of shape (n_samples, n_features).") @@ -120,7 +122,8 @@ def _check_array(self, X, *, reset): self.n_features_in_ = X.shape[1] elif X.shape[1] != self.n_features_in_: raise PretabDataError( - f"Input has {X.shape[1]} features, but NumericBinningTransformer was fitted with {self.n_features_in_}." + f"X has {X.shape[1]} features, but NumericBinningTransformer is expecting " + f"{self.n_features_in_} features as input." ) return X @@ -250,8 +253,7 @@ def get_feature_names_out(self, input_features=None): ``"{feature}_bin{k}"`` name per bin. """ check_is_fitted(self, "n_features_in_") - if input_features is None: - input_features = [f"x{i}" for i in range(self.n_features_in_)] + input_features = resolve_input_features(self, input_features) if self.encode == "ordinal": return np.asarray(input_features, dtype=object) check_is_fitted(self, "n_bins_") diff --git a/pretab/encoding/numerical/ple.py b/pretab/encoding/numerical/ple.py index cca1e86..8594bd6 100644 --- a/pretab/encoding/numerical/ple.py +++ b/pretab/encoding/numerical/ple.py @@ -10,12 +10,13 @@ import numpy as np from sklearn.base import BaseEstimator, TransformerMixin -from sklearn.utils.validation import check_array, check_is_fitted +from sklearn.utils.validation import _check_feature_names, check_array, check_is_fitted from ...core.adaptive import AdaptiveResolutionMixin from ...core.parameters import UNSET, AliasResolverMixin from ...core.representation import RepresentationSpecMixin from ...core.supervised import warn_target_leakage +from ...core.validation import resolve_input_features from ...exceptions import ( IncompatibleParamsError, InvalidParamError, @@ -161,6 +162,7 @@ def fit(self, X, y=None): "PLETransformer is always target-aware and requires y at fit time; got y=None." ) + _check_feature_names(self, X, reset=True) X = check_array( X, dtype=np.float64, # type: ignore @@ -230,6 +232,7 @@ def transform(self, X): """ check_is_fitted(self, ["thresholds_", "n_features_in_"]) + _check_feature_names(self, X, reset=False) X = check_array( X, dtype=np.float64, # type: ignore @@ -326,9 +329,7 @@ def get_feature_names_out(self, input_features=None): One name per output column, formatted ``{name}_ple{j}``. """ check_is_fitted(self, ["thresholds_", "n_features_in_"]) - - if input_features is None: - input_features = [f"x{i}" for i in range(self.n_features_in_)] + input_features = resolve_input_features(self, input_features) feature_names_out = [] for name, n_bins in zip(input_features, self.n_bins_per_feature_, strict=False): diff --git a/pretab/expansion/functional/base.py b/pretab/expansion/functional/base.py index 95432f9..81ce176 100644 --- a/pretab/expansion/functional/base.py +++ b/pretab/expansion/functional/base.py @@ -117,9 +117,17 @@ def fit(self, X, y=None): for i in range(X.shape[1]): if np.isnan(X[:, i]).all(): raise PretabDataError(f"Feature at index {i} has only NaN values") - self.centers_.append(adapter.get_centers(X[:, i], y_place, min_centers, max_centers)) + if self.target_aware: + centers = adapter.get_centers(X[:, i], y_place, min_centers, max_centers) + else: + centers = self._unsupervised_centers(adapter, X[:, i], n_centers) + self.centers_.append(centers) return self + def _unsupervised_centers(self, adapter, x, n_centers): + """Return ``n_centers`` quantile / uniform centers spanning the feature range.""" + return adapter.get_centers(x, None, n_centers, n_centers) + def transform(self, X): """Expand every feature against its centers and stack the results.""" check_is_fitted(self, "centers_") diff --git a/pretab/expansion/functional/relu.py b/pretab/expansion/functional/relu.py index 6b2496a..e6b3f64 100644 --- a/pretab/expansion/functional/relu.py +++ b/pretab/expansion/functional/relu.py @@ -78,6 +78,16 @@ class ReLUExpansionTransformer(BaseCenterExpansion): _feature_suffix_value = "relu" _representation_family = "relu" + def _unsupervised_centers(self, adapter, x, n_centers): + """Return ``n_centers`` centers that each start a ramp inside the data. + + A ramp ``max(0, x - c)`` centered at the training maximum is identically + zero on the training data, so ``n_centers + 1`` range-spanning locations + are placed and the right endpoint is dropped (the first center stays at + the minimum, giving the linear term). + """ + return adapter.get_centers(x, None, n_centers + 1, n_centers + 1)[:-1] + def __init__( self, output_dim=UNSET, diff --git a/pretab/expansion/spline/b_spline.py b/pretab/expansion/spline/b_spline.py index 293471c..44abe1f 100644 --- a/pretab/expansion/spline/b_spline.py +++ b/pretab/expansion/spline/b_spline.py @@ -81,6 +81,8 @@ def _design_matrix(self, x: np.ndarray, knots: np.ndarray) -> np.ndarray: for i in range(n_basis): coef = np.zeros(n_basis) coef[i] = 1.0 - spline = BSpline(knots, coef, self.degree, extrapolate=False) + # In-range values are the same either way; out-of-range values only reach + # this point under an "extrapolate" / "warn" policy (the default clips). + spline = BSpline(knots, coef, self.degree, extrapolate=True) design[:, i] = spline(x) return np.nan_to_num(design, nan=0.0) diff --git a/pretab/expansion/spline/base.py b/pretab/expansion/spline/base.py index a6cfa61..73cfa64 100644 --- a/pretab/expansion/spline/base.py +++ b/pretab/expansion/spline/base.py @@ -14,6 +14,7 @@ basis matrix through the ``_design_matrix`` hook. """ +import warnings from typing import ClassVar, Literal import numpy as np @@ -30,9 +31,10 @@ from ...core.policy import RepresentationPolicy, resolve_out_of_range from ...core.supervised import warn_target_leakage from ...exceptions import ( - IncompatibleParamsError, + DataWarning, InvalidParamError, PretabDataError, + invalid_param_error, ) from ...placement.adapters import SplinePlacementAdapter @@ -64,7 +66,10 @@ class BaseSplineTransformer(BasePreTabTransformer): knot_locations : ndarray or None, default=None Explicit internal knot locations applied to every feature. Takes priority over ``target_aware`` placement and the automatic strategy, and overrides - ``output_dim``. + ``output_dim`` and the adaptive bounds: a feature gets + ``len(knot_locations) + degree + 1`` basis functions. Repeated knots are + merged; knots on or outside a feature's fitted range are dropped for that + feature with a :class:`~pretab.exceptions.DataWarning`. target_aware : bool, default=False If True, knots are placed by a target-aware selector built from @@ -222,23 +227,22 @@ def _column_knots( max_basis_req: int | None, ) -> np.ndarray: """Build the full knot vector for a single feature.""" + x_min = x_valid.min() + x_max = x_valid.max() + boundary_left = np.repeat(x_min, self.degree + 1) + boundary_right = np.repeat(x_max, self.degree + 1) + + if self.knot_locations is not None: + internal_knots = self._explicit_knots(x_min, x_max) + return np.concatenate([boundary_left, internal_knots, boundary_right]) + min_basis, max_basis = self._resolve_basis_bounds(n_basis, min_basis_req, max_basis_req) min_knots = self._basis_to_knots(min_basis) max_knots = self._basis_to_knots(max_basis) if self.adaptive: max_knots = min(max_knots, max(0, np.unique(x_valid).size - 2)) - x_min = x_valid.min() - x_max = x_valid.max() - - if self.knot_locations is not None: - expected_knots = self._basis_to_knots(n_basis) - if not self.adaptive and len(self.knot_locations) != expected_knots: - raise IncompatibleParamsError( - "knot_locations length must match output_dim - degree - 1 when adaptive=False" - ) - internal_knots = self._adjust_internal_knots(x_valid, np.asarray(self.knot_locations), min_knots, max_knots) - elif selector is not None: + if selector is not None: # Search exactly the window this feature needs, so the selector's own # importance ranking picks the knots instead of a positional trim. min_knots = min(min_knots, max_knots) @@ -251,10 +255,41 @@ def _column_knots( internal_knots = self._generate_knots(x_valid, n_internal, strategy) internal_knots = self._adjust_internal_knots(x_valid, internal_knots, min_knots, max_knots) - boundary_left = np.repeat(x_min, self.degree + 1) - boundary_right = np.repeat(x_max, self.degree + 1) return np.concatenate([boundary_left, internal_knots, boundary_right]) + def _explicit_knots(self, x_min: float, x_max: float) -> np.ndarray: + """Return the user's ``knot_locations`` that are valid interior knots for one feature. + + Repeated knots are merged. A knot on or beyond the fitted range would give + a basis function with zero-width support, so such knots are dropped with + a :class:`~pretab.exceptions.DataWarning` instead of being replaced. + """ + requested = np.unique(np.asarray(self.knot_locations, dtype=float)) + inside = (requested > x_min) & (requested < x_max) + if not inside.all(): + warnings.warn( + f"knot_locations {requested[~inside].tolist()} lie outside the open feature range " + f"({float(x_min)!r}, {float(x_max)!r}) and were dropped: a knot on or beyond the " + "boundary gives a basis function with zero-width support.", + DataWarning, + stacklevel=4, + ) + return requested[inside] + + def _validate_knot_locations(self) -> None: + """Reject ``knot_locations`` that are not a 1-D sequence of finite numbers.""" + try: + knots = np.asarray(self.knot_locations, dtype=float) + except (TypeError, ValueError): + knots = None + if knots is None or knots.ndim != 1 or not np.isfinite(knots).all(): + raise invalid_param_error( + type(self).__name__, + "knot_locations", + self.knot_locations, + "must be a 1-D sequence of finite numbers (the interior knots shared by every feature)", + ) + def fit(self, X, y=None): """Determine per-feature knot vectors.""" warn_target_leakage(self, y) @@ -262,9 +297,12 @@ def fit(self, X, y=None): n_basis = self._resolve_param("output_dim", default=6) min_basis_req = self._resolve_param("min_output_dim", default=None) max_basis_req = self._resolve_param("max_output_dim", default=None) - if n_basis < self.degree + 1: + if self.knot_locations is not None: + # Explicit knots fix the width; output_dim and the adaptive bounds are ignored. + self._validate_knot_locations() + elif n_basis < self.degree + 1: raise InvalidParamError(f"output_dim must be >= degree + 1 = {self.degree + 1}, got {n_basis}") - if n_basis > 50: + elif n_basis > 50: raise InvalidParamError(f"output_dim should be <= 50, got {n_basis}") X = self._validate(X, reset=True) diff --git a/pretab/expansion/spline/m_spline.py b/pretab/expansion/spline/m_spline.py index beac100..101b019 100644 --- a/pretab/expansion/spline/m_spline.py +++ b/pretab/expansion/spline/m_spline.py @@ -85,7 +85,9 @@ def _mspline_basis(self, x: np.ndarray, knots: np.ndarray, basis_idx: int) -> np n_coef = len(knots) - self.degree - 1 coef = np.zeros(n_coef) coef[basis_idx] = 1.0 - spline = BSpline(knots, coef, self.degree, extrapolate=False) + # In-range values are the same either way; out-of-range values only reach + # this point under an "extrapolate" / "warn" policy (the default clips). + spline = BSpline(knots, coef, self.degree, extrapolate=True) values = np.nan_to_num(spline(x), nan=0.0) knot_span = knots[basis_idx + self.degree + 1] - knots[basis_idx] diff --git a/pretab/expansion/spline/multivariate/tensor_product.py b/pretab/expansion/spline/multivariate/tensor_product.py index 2be256b..d862ec2 100644 --- a/pretab/expansion/spline/multivariate/tensor_product.py +++ b/pretab/expansion/spline/multivariate/tensor_product.py @@ -2,9 +2,10 @@ from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted -from ....core.knots import bspline_basis +from ....core.knots import bspline_basis, extrapolate_bspline_rows from ....core.parameters import UNSET from ....core.policy import RepresentationPolicy, resolve_out_of_range +from ....core.validation import resolve_input_features from ....exceptions import InvalidParamError from ..mixins import SplineBasisMixin @@ -165,6 +166,7 @@ def _basis_matrix(self, x, knots): B = np.zeros((len(x), n_basis)) for i in range(n_basis): B[:, i] = bspline_basis(x, knots, self.degree, i) + B = extrapolate_bspline_rows(B, x, knots, self.degree) if self.include_bias: B = np.hstack([np.ones((len(x), 1)), B]) return B @@ -246,8 +248,7 @@ def transform(self, X): def get_feature_names_out(self, input_features=None): """Return names for the interaction basis as ``tp_{feat0 i}_{feat1 j}...``.""" check_is_fitted(self, "marginal_sizes_") - if input_features is None: - input_features = [f"x{i}" for i in range(self.n_features_in_)] + input_features = resolve_input_features(self, input_features) sizes = self.marginal_sizes_ names = [] for multi_index in np.ndindex(*sizes): diff --git a/pretab/expansion/spline/p_spline.py b/pretab/expansion/spline/p_spline.py index 8328484..5ef6152 100644 --- a/pretab/expansion/spline/p_spline.py +++ b/pretab/expansion/spline/p_spline.py @@ -2,7 +2,7 @@ from sklearn.base import BaseEstimator, TransformerMixin from sklearn.utils.validation import check_is_fitted -from ...core.knots import bspline_basis +from ...core.knots import bspline_basis, extrapolate_bspline_rows from ...core.parameters import UNSET from ...core.policy import RepresentationPolicy, resolve_out_of_range from ...exceptions import InvalidParamError @@ -210,6 +210,7 @@ def transform(self, X): basis = np.zeros((len(x), nb)) for j in range(nb): basis[:, j] = bspline_basis(x, knots, self.degree, j) + basis = extrapolate_bspline_rows(basis, x, knots, self.degree) if self.include_bias: basis = np.hstack([np.ones((len(x), 1)), basis]) all_basis.append(basis) diff --git a/pretab/placement/unsupervised.py b/pretab/placement/unsupervised.py index 1ba3000..53c5fb6 100644 --- a/pretab/placement/unsupervised.py +++ b/pretab/placement/unsupervised.py @@ -24,7 +24,7 @@ import numpy as np -from ..core.knots import quantile_knots, spanning_knots, uniform_knots +from ..core.knots import quantile_knots, spanning_knots, supplement_interior_knots, uniform_knots from .base import BasePlacementStrategy, PlacementResult __all__ = ["QuantilePlacement", "UniformPlacement"] @@ -80,11 +80,26 @@ def _place(self, x: np.ndarray, n_units: int) -> np.ndarray: class QuantilePlacement(_UnsupervisedPlacement): - """Locations at evenly spaced data quantiles of a feature.""" + """Locations at evenly spaced data quantiles of a feature. + + On tied data (zero-inflated, top-coded or discrete features) several quantiles + coincide, which would repeat a location and duplicate an output column. The + locations are therefore made distinct: the endpoints (when included) are kept + once and the remaining locations are topped up with strictly interior + quantile / uniform candidates (see + :func:`pretab.core.knots.supplement_interior_knots`). A zero-range feature, + which has no interior, keeps its repeated location so the count still holds. + """ name: ClassVar[str] = "quantile" def _place(self, x: np.ndarray, n_units: int) -> np.ndarray: - if self.include_endpoints: - return spanning_knots(x, n_units, "quantile") - return quantile_knots(x, n_units) + if x.size == 0 or x.max() <= x.min(): + return spanning_knots(x, n_units, "quantile") if self.include_endpoints else quantile_knots(x, n_units) + if not self.include_endpoints: + return supplement_interior_knots(x, quantile_knots(x, n_units), n_units) + locations = spanning_knots(x, n_units, "quantile") + if n_units <= 2: + return locations + interior = supplement_interior_knots(x, locations[1:-1], n_units - 2) + return np.concatenate([[x.min()], interior, [x.max()]]) diff --git a/pretab/preprocessor.py b/pretab/preprocessor.py index 5be94d2..d11511b 100644 --- a/pretab/preprocessor.py +++ b/pretab/preprocessor.py @@ -1,10 +1,11 @@ +import copy import hashlib import json import logging import os import time import warnings -from typing import cast +from typing import Any, cast import numpy as np from scipy import sparse as sp @@ -22,11 +23,13 @@ ) from .compose.inspection import ( block_name, + block_uses_target, build_feature_info, build_feature_lineage, build_transformer_summary, feature_names_out, get_output_slices, + representation_leaf, ) from .compose.output import ( compute_output_report, @@ -102,9 +105,13 @@ def _preset_numerical_method(task) -> str: def _spec_json(spec: dict, **kwargs) -> str: - """Serialize a spec to JSON text, raising a typed error for unsupported state.""" + """Serialize a spec to strict JSON text, raising a typed error for unsupported state. + + Non-finite floats are tagged by the encoder, so ``allow_nan=False`` keeps bare + ``NaN`` / ``Infinity`` tokens (rejected by strict JSON parsers) out of the spec. + """ try: - return json.dumps(spec, **kwargs) + return json.dumps(spec, allow_nan=False, **kwargs) except (TypeError, ValueError) as exc: raise PretabSerializationError(f"The fitted state cannot be written as JSON: {exc}") from exc @@ -604,6 +611,12 @@ def fit(self, X, y=None, embeddings=None): categorical_features, sparse_threshold=sparse_threshold, ) + # The Preprocessor wraps its own output (set_output / output_format), so the + # internal steps always produce plain arrays. Otherwise a global + # ``sklearn.set_config(transform_output="pandas")`` reaches them: the sparse + # OneHotEncoder refuses pandas output, and mixed float / bool indicator + # blocks are stacked into an object-dtype frame. + self.column_transformer_.set_output(transform="default") # ColumnTransformer.fit runs fit_transform internally; keep the training # output to resolve output_format="auto" once, from the training density. training_output = self.column_transformer_.fit_transform(self._column_transformer_input(X), y) @@ -668,15 +681,7 @@ def transform(self, X, embeddings=None, return_array=None): check_is_fitted(self) - # Array columns are matched by position, so a different width can never be - # routed correctly; an extra column would silently shift or drop features. - if isinstance(X, np.ndarray) and X.ndim == 2 and X.shape[1] != self.n_features_in_: - raise PretabDataError( - f"X has {X.shape[1]} features, but {type(self).__name__} is expecting " - f"{self.n_features_in_} features as input." - ) - - X = to_dataframe(X, copy=True) + X = self._align_input(X) if self.missing_policy == "error": self._reject_missing(X) @@ -881,6 +886,85 @@ def output_dims_(self) -> dict: dims[self._input_label(columns[0])] = width return dims + def _align_input(self, X): + """Return ``X`` as a DataFrame whose columns line up with the fitted columns. + + A NumPy array is matched by position. Once the preprocessor was fitted on + named columns, an array of the fitted width takes those column labels + (numeric columns of an object array are re-inferred), and a DataFrame + passed to a preprocessor fitted on an array is matched by position too -- + both with scikit-learn's usual warning about the missing or unexpected + feature names. Any other DataFrame is matched by column label. + + Raises + ------ + PretabDataError + If array-like input matched by position has a different number of + columns than seen during ``fit``: it could never be routed correctly. + """ + fitted_labels = getattr(self.column_transformer_, "feature_names_in_", None) + positional_labels = [f"feature_{i}" for i in range(self.n_features_in_)] + fitted_on_array = ( + not hasattr(self, "feature_names_in_") + and fitted_labels is not None + and list(fitted_labels) == positional_labels + ) + + is_array = isinstance(X, np.ndarray) + X = to_dataframe(X, copy=True) + if not (is_array or (fitted_on_array and list(X.columns) != positional_labels)): + return X + if X.shape[1] != self.n_features_in_: + raise PretabDataError( + f"X has {X.shape[1]} features, but {type(self).__name__} is expecting " + f"{self.n_features_in_} features as input." + ) + if fitted_labels is None or list(X.columns) == list(fitted_labels): + return X + if hasattr(self, "feature_names_in_"): + warnings.warn( + f"X does not have valid feature names, but {type(self).__name__} was fitted with feature names", + UserWarning, + stacklevel=3, + ) + elif not is_array: + warnings.warn( + f"X has feature names, but {type(self).__name__} was fitted without feature names", + UserWarning, + stacklevel=3, + ) + X = X.set_axis(list(fitted_labels), axis=1) + return X.infer_objects() if is_array else X + + def _cross_fit_fold(self, X, y): + """Return a copy of this fit whose target-aware blocks are refit on ``(X, y)``. + + Used by :class:`~pretab.CrossFittedTransformer` for its out-of-fold + features. Only the blocks that consume the target are refit on the fold; + column types, category vocabularies and every other block are shared + with this all-data fit, so an out-of-fold row is encoded exactly like + ``transform`` encodes it, except that no target-aware placement has seen + that row's target. Refitting the whole preprocessor per fold would + re-detect column types and relearn categories on fewer rows, shifting + integer codes and one-hot columns between the folds and ``transform``. + """ + check_is_fitted(self) + X_ct = self._column_transformer_input(self._align_input(X)) + column_transformer = copy.copy(self.column_transformer_) + column_transformer.transformers_ = [ + ( + name, + cast(Any, clone(transformer)).fit(X_ct[list(columns)], y) + if name != "remainder" and block_uses_target(transformer, columns) + else transformer, + columns, + ) + for name, transformer, columns in self.column_transformer_.transformers_ + ] + fold = copy.copy(self) + fold.column_transformer_ = column_transformer + return fold + @staticmethod def _column_transformer_input(X): """Adapt a frame to what the internal ColumnTransformer expects. @@ -1065,7 +1149,7 @@ def _log_internal_decisions(self): """Log fitted internal decisions (bins / knots / centers) at DEBUG.""" for step_name, transformer, columns in self.column_transformer_.transformers_: name = block_name(step_name, columns) - last_step = transformer.steps[-1][1] if hasattr(transformer, "steps") else transformer + last_step = representation_leaf(transformer) for attr in ( "thresholds_", "knots_", diff --git a/tests/compose/test_inspection.py b/tests/compose/test_inspection.py index 1921e62..0d4f950 100644 --- a/tests/compose/test_inspection.py +++ b/tests/compose/test_inspection.py @@ -1,5 +1,7 @@ """Unit tests for :mod:`pretab.compose.inspection`.""" +import warnings + import numpy as np import pandas as pd import pytest @@ -56,6 +58,23 @@ def test_build_feature_info_does_not_infer_kind_from_cat_in_feature_name(make_co assert feature not in categorical +@pytest.mark.parametrize(("method", "dimension"), [("rbf", 3), ("standardization", 1)]) +def test_build_feature_info_probes_a_step_fitted_on_the_frame_without_warning( + make_config, sample_frame, method, dimension +): + """Without an imputer the representation is fitted on the DataFrame column and + records ``feature_names_in_``; probing it with a bare array warned that X does + not have valid feature names.""" + config = make_config(numerical_method=method, numerical_imputation=None, output_dim=3) + ct = build_column_transformer(config, ["age"], ["city"]).fit(sample_frame) + + with warnings.catch_warnings(): + warnings.simplefilter("error") + numerical, _, _ = build_feature_info(ct, embeddings=False, embedding_dimensions={}) + + assert numerical["age"]["dimension"] == dimension + + def test_build_transformer_summary_has_header_and_rows(): numerical = {"age": {"preprocessing": "imputer -> standardization", "dimension": 1, "categories": None}} categorical = {"city": {"preprocessing": "imputer -> continuous_ordinal", "dimension": 1, "categories": 3}} diff --git a/tests/compose/test_search.py b/tests/compose/test_search.py index 7268892..c3e065f 100644 --- a/tests/compose/test_search.py +++ b/tests/compose/test_search.py @@ -124,3 +124,33 @@ def test_get_params_and_clone_preserve_config(): cloned = clone(search) assert isinstance(cloned, RepresentationSearchCV) assert cloned.get_params()["cv"] == 3 + + +@pytest.fixture +def class_data(): + rng = np.random.default_rng(0) + X = pd.DataFrame({"x": rng.uniform(-3, 3, 300)}) + return X, np.where(np.sin(X["x"]) > 0, "pos", "neg") + + +def test_classifier_search_places_against_class_labels(class_data): + """A classifier made the search keep task="regression": string labels crashed the + target-aware placement and integer labels were treated as a regression target.""" + X, y = class_data + search = RepresentationSearchCV(LogisticRegression(), methods=["minmax", "ple"], cv=3, random_state=0).fit(X, y) + assert search.best_preprocessor_.task == "classification" + assert set(search.predict(X)) <= {"pos", "neg"} + + +def test_classifier_search_keeps_an_explicit_task(class_data): + X, y = class_data + search = RepresentationSearchCV( + LogisticRegression(), methods=["minmax"], cv=3, preprocessor_params={"task": "regression"} + ).fit(X, (y == "pos").astype(int)) + assert search.best_preprocessor_.task == "regression" + + +def test_regressor_search_keeps_the_regression_task(nonlinear_data): + X, y = nonlinear_data + search = RepresentationSearchCV(LinearRegression(), methods=["minmax"], cv=3).fit(X, y) + assert search.best_preprocessor_.task == "regression" diff --git a/tests/core/test_cross_fitted.py b/tests/core/test_cross_fitted.py index f39ddda..515ed70 100644 --- a/tests/core/test_cross_fitted.py +++ b/tests/core/test_cross_fitted.py @@ -164,8 +164,115 @@ def test_wrapped_column_transformer_selects_by_column_name(mixed_frame): assert out.shape == (300, 4) +def test_spec_after_a_dataframe_fit_names_the_columns(mixed_frame): + """The default spec passed x0, x1, ... to get_feature_names_out, which a + transformer fitted on a DataFrame rejects as not equal to feature_names_in_.""" + X, y = mixed_frame + cf = CrossFittedTransformer(PLETransformer(output_dim=4), n_folds=3, random_state=0).fit(X[["num"]], y) + spec = cf.get_representation_spec() + + assert spec.input_features == ("num",) + assert spec.output_features == tuple(cf.get_feature_names_out()) + assert spec.cross_fitted is True + + +def test_spec_of_a_wrapped_preprocessor_names_the_columns(mixed_frame): + """A wrapped estimator without its own spec takes the fallback path, which + used to fail the same way for a Preprocessor fitted on a DataFrame.""" + from pretab import Preprocessor + + X, y = mixed_frame + cf = CrossFittedTransformer(Preprocessor(output_dim=6, random_state=0), n_folds=3, random_state=0).fit(X, y) + spec = cf.get_representation_spec() + + assert spec.input_features == ("num", "city") + assert spec.output_features == tuple(cf.get_feature_names_out()) + assert (spec.cross_fitted, spec.n_folds) == (True, 3) + + def test_one_dimensional_input_is_still_a_single_column(): rng = np.random.default_rng(0) x = rng.normal(size=200) out = CrossFittedTransformer(PLETransformer(output_dim=4), n_folds=3, random_state=0).fit_transform(x, x**2) assert out.shape == (200, 4) + + +# --- a wrapped Preprocessor keeps one encoding across folds ------------------------- + + +@pytest.fixture +def frame_with_rare_category(): + import pandas as pd + + rng = np.random.default_rng(0) + X = pd.DataFrame( + { + "x": rng.normal(size=300), + "rooms": rng.integers(1, 9, 300), # unique ratio near the default cat_cutoff + "city": rng.choice(["Berlin", "Paris", "Rome"], 300), + } + ) + X.loc[7, "city"] = "Amsterdam" # a category some folds never see + y = X["x"].to_numpy() ** 2 + (X["city"] == "Paris") + rng.normal(scale=0.1, size=300) + return X, y + + +def _split_columns(cross_fitted, prefix): + names = list(cross_fitted.get_feature_names_out()) + return [i for i, name in enumerate(names) if name.startswith(prefix)] + + +def test_wrapped_preprocessor_out_of_fold_codes_match_transform(frame_with_rare_category): + """Each fold used to refit the whole Preprocessor: category codes shifted when a + category was missing from a fold, and column types were re-detected on fewer rows.""" + from pretab import Preprocessor + + X, y = frame_with_rare_category + cross_fitted = CrossFittedTransformer(Preprocessor(output_dim=4, random_state=0), n_folds=5, random_state=0) + out = cross_fitted.fit_transform(X, y) + full = np.asarray(cross_fitted.transform(X)) + + unsupervised = _split_columns(cross_fitted, "cat_") + assert unsupervised # rooms and city are categorical on the full data + np.testing.assert_array_equal(out[:, unsupervised], full[:, unsupervised]) + + +def test_wrapped_preprocessor_target_aware_blocks_stay_out_of_fold(frame_with_rare_category): + from pretab import Preprocessor + + X, y = frame_with_rare_category + cross_fitted = CrossFittedTransformer(Preprocessor(output_dim=4, random_state=0), n_folds=5, random_state=0) + out = cross_fitted.fit_transform(X, y) + supervised = _split_columns(cross_fitted, "num_x_ple") + assert not np.allclose(out[:, supervised], np.asarray(cross_fitted.transform(X))[:, supervised]) + + +def test_wrapped_preprocessor_with_one_hot_keeps_a_fixed_width(frame_with_rare_category): + from pretab import Preprocessor + + X, y = frame_with_rare_category + wrapped = Preprocessor(categorical_method="one-hot", output_dim=4, random_state=0) + cross_fitted = CrossFittedTransformer(wrapped, n_folds=5, random_state=0) + out = cross_fitted.fit_transform(X, y) + assert out.shape == (len(X), len(cross_fitted.get_feature_names_out())) + + +def test_wrapped_unsupervised_preprocessor_matches_transform(frame_with_rare_category): + from pretab import Preprocessor + + X, y = frame_with_rare_category + wrapped = Preprocessor(numerical_method="minmax", target_aware=False, placement_strategy="quantile") + cross_fitted = CrossFittedTransformer(wrapped, n_folds=3, random_state=0) + np.testing.assert_array_equal(cross_fitted.fit_transform(X, y), np.asarray(cross_fitted.transform(X))) + + +def test_cross_fitting_densifies_sparse_output_and_rejects_blocks(frame_with_rare_category): + from pretab import Preprocessor + + X, y = frame_with_rare_category + sparse = Preprocessor(categorical_method="one-hot", output_format="sparse", random_state=0) + out = CrossFittedTransformer(sparse, n_folds=3, random_state=0).fit_transform(X, y) + assert isinstance(out, np.ndarray) + blocks = Preprocessor(output_structure="blocks", random_state=0) + with pytest.raises(IncompatibleParamsError, match="dict of blocks"): + CrossFittedTransformer(blocks, n_folds=3, random_state=0).fit_transform(X, y) diff --git a/tests/core/test_policy.py b/tests/core/test_policy.py index 0b99b20..7d4abbe 100644 --- a/tests/core/test_policy.py +++ b/tests/core/test_policy.py @@ -14,6 +14,8 @@ this pass. """ +import warnings + import numpy as np import pytest @@ -125,3 +127,63 @@ def test_policy_dict_is_accepted(): transformer = BSplineTransformer(output_dim=8, policy={"out_of_range": "error"}).fit(X) with pytest.raises(PretabDataError): transformer.transform(np.array([[20.0]])) + + +# --- partial overrides and the "extrapolate" / "warn" reactions --------------------- + +_CLIP_FAMILIES = [ + (BSplineTransformer, {"output_dim": 6}), + (MSplineTransformer, {"output_dim": 6}), + (ISplineTransformer, {"output_dim": 6}), + (PSplineTransformer, {}), +] + + +@pytest.mark.parametrize(("cls", "kwargs"), _CLIP_FAMILIES) +def test_partial_policy_mapping_keeps_the_family_out_of_range_default(cls, kwargs): + """A mapping naming only "constant" used to reset out_of_range to the dataclass + default "extrapolate", silently turning the clip families' out-of-range rows to zero.""" + X = np.linspace(0, 1, 50).reshape(-1, 1) + probe = np.array([[1.05], [-0.05]]) + default = cls(**kwargs).fit(X).transform(probe) + partial = cls(policy={"constant": "warn"}, **kwargs).fit(X).transform(probe) + np.testing.assert_array_equal(partial, default) + + +def test_tensorproduct_partial_policy_mapping_keeps_clipping(): + X = np.random.default_rng(0).uniform(0, 1, (200, 2)) + probe = np.array([[1.05, 0.5]]) + default = TensorProductSplineTransformer().fit(X).transform(probe) + partial = TensorProductSplineTransformer(policy={"constant": "warn"}).fit(X).transform(probe) + np.testing.assert_array_equal(partial, default) + + +def test_policy_mapping_with_an_unknown_axis_is_rejected(): + from pretab.exceptions import InvalidParamError + + with pytest.raises(InvalidParamError): + BSplineTransformer(policy={"missing": "error"}).fit(np.linspace(0, 1, 50).reshape(-1, 1)).transform([[0.5]]) + + +@pytest.mark.parametrize("cls", [BSplineTransformer, PSplineTransformer]) +@pytest.mark.parametrize("reaction", ["extrapolate", "warn"]) +def test_extrapolated_bspline_rows_are_not_zeroed(cls, reaction): + """Out-of-range rows were all zero under "extrapolate" / "warn": the basis is zero + outside the knot span. They now extend the boundary pieces, a partition of unity.""" + X = np.linspace(0, 1, 50).reshape(-1, 1) + transformer = cls(policy=RepresentationPolicy(out_of_range=reaction)).fit(X) + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DataWarning) + out = transformer.transform(np.array([[1.05], [-0.05], [0.5]])) + np.testing.assert_allclose(out.sum(axis=1), 1.0) + clipped = cls().fit(X).transform(np.array([[0.5]])) + np.testing.assert_array_equal(out[2], clipped[0]) # in-range rows are unchanged + + +def test_extrapolated_mspline_and_tensor_rows_are_not_zeroed(): + policy = RepresentationPolicy(out_of_range="extrapolate") + X = np.linspace(0, 1, 50).reshape(-1, 1) + assert (MSplineTransformer(output_dim=6, policy=policy).fit(X).transform([[1.05]]).sum() > 0).all() + X2 = np.random.default_rng(0).uniform(0, 1, (200, 2)) + tensor = TensorProductSplineTransformer(policy=policy).fit(X2) + np.testing.assert_allclose(tensor.transform([[1.05, 0.5]]).sum(), 1.0) diff --git a/tests/core/test_representation_spec.py b/tests/core/test_representation_spec.py index 592a098..f444e7b 100644 --- a/tests/core/test_representation_spec.py +++ b/tests/core/test_representation_spec.py @@ -1,8 +1,11 @@ import warnings import numpy as np +import pandas as pd import pytest +from sklearn.base import clone +from pretab import transformers from pretab.core.representation import RepresentationSpec, RepresentationSpecMixin from pretab.transformers import ( BSplineTransformer, @@ -203,6 +206,32 @@ def test_every_transformer_has_representation_spec(): assert hasattr(transformer, "get_representation_spec") +def test_cases_cover_every_exported_representation(): + exported = {getattr(transformers, name) for name in transformers.__all__} + with_spec = {cls for cls in exported if hasattr(cls, "get_representation_spec")} + assert with_spec == {type(case[1]) for case in CASES} + + +@pytest.mark.parametrize( + ("transformer", "X", "y"), + [(case[1], case[2], case[3]) for case in CASES], + ids=CASE_IDS, +) +def test_spec_after_a_dataframe_fit_follows_get_feature_names_out(transformer, X, y): + """A DataFrame fit records ``feature_names_in_``; the default spec passed + ``x0, x1, ...`` to ``get_feature_names_out``, which then raised a ValueError. + The spec now resolves its input names exactly like ``get_feature_names_out``.""" + frame = pd.DataFrame({f"feat_{i}": X[:, i] for i in range(X.shape[1])}) + fitted = _fit(clone(transformer), frame, y) + spec = fitted.get_representation_spec() + + generic = [f"x{i}" for i in range(X.shape[1])] + inputs = [str(name) for name in getattr(fitted, "feature_names_in_", generic)] + assert spec.input_features == tuple(inputs) + assert spec.output_features == tuple(str(name) for name in fitted.get_feature_names_out(inputs)) + assert fitted.get_representation_spec(input_features=inputs) == spec + + def test_periodic_spec_reports_period(): x_month = RNG.uniform(0.0, 12.0, size=(80, 1)) transformer = _fit(PeriodicEncodingTransformer(period=12, harmonics=1), x_month) diff --git a/tests/expansion/functional/test_reluexpansion_transformer.py b/tests/expansion/functional/test_reluexpansion_transformer.py index 8004c5f..7430410 100644 --- a/tests/expansion/functional/test_reluexpansion_transformer.py +++ b/tests/expansion/functional/test_reluexpansion_transformer.py @@ -70,3 +70,22 @@ def test_relu_feature_mismatch(X_multi_feature, y_regression): transformer.fit(X_multi_feature[:, :2], y_regression) with pytest.raises(ValueError, match="is expecting"): transformer.transform(X_multi_feature) + + +@pytest.mark.parametrize("placement_strategy", ["uniform", "quantile"]) +def test_unsupervised_relu_has_no_dead_column(placement_strategy): + """The last center sat at the training maximum, so max(0, x - c) was identically + zero on the training data and only output_dim - 1 columns carried signal.""" + X = np.random.default_rng(0).uniform(0, 10, (500, 1)) + transformer = ReLUExpansionTransformer(output_dim=6, placement_strategy=placement_strategy).fit(X) + out = transformer.transform(X) + assert out.shape == (500, 6) + assert (out.max(axis=0) > 0).all() + assert np.linalg.matrix_rank(out) == 6 + assert transformer.centers_[0][0] == X.min() + assert transformer.centers_[0][-1] < X.max() + + +def test_unsupervised_relu_keeps_its_width_on_a_constant_feature(): + X = np.full((20, 1), 3.0) + assert ReLUExpansionTransformer(output_dim=4).fit(X).transform(X).shape == (20, 4) diff --git a/tests/expansion/spline/test_spline_expansions.py b/tests/expansion/spline/test_spline_expansions.py index 3ab0b88..4416d83 100644 --- a/tests/expansion/spline/test_spline_expansions.py +++ b/tests/expansion/spline/test_spline_expansions.py @@ -253,3 +253,41 @@ def test_adaptive_spline_width_is_not_capped_at_the_legacy_window(cls): adaptive=True, min_output_dim=5, max_output_dim=40, target_aware=True, placement_strategy="cart" ).fit(x.reshape(-1, 1), y) assert 15 < transformer.total_output_dim_ <= 40 + + +# --- explicit knot_locations ---------------------------------------------------------- + + +@pytest.mark.parametrize("cls", [BSplineTransformer, MSplineTransformer, ISplineTransformer]) +@pytest.mark.parametrize("knots", [[5.0], [2.0, 4.0, 6.0, 8.0], list(np.linspace(1, 9, 12))]) +def test_knot_locations_set_the_width_regardless_of_output_dim(cls, knots): + """knot_locations raised unless its length matched the default output_dim, although + the docstring says output_dim is ignored when knots are given.""" + X = np.linspace(0, 10, 50).reshape(-1, 1) + transformer = cls(knot_locations=knots).fit(X) + np.testing.assert_allclose(transformer.knots_[0][4:-4], knots) + assert transformer.transform(X).shape == (50, len(knots) + 4) + + +def test_knot_locations_are_kept_in_adaptive_mode(): + X = np.linspace(0, 10, 50).reshape(-1, 1) + knots = np.array([2.0, 4.0, 6.0, 8.0]) + transformer = BSplineTransformer(knot_locations=knots, adaptive=True, min_output_dim=5, max_output_dim=6) + np.testing.assert_array_equal(transformer.fit(X).knots_[0][4:-4], [2, 4, 6, 8]) + + +def test_knot_locations_outside_the_range_are_dropped_with_a_warning(): + from pretab.exceptions import DataWarning + + X = np.linspace(0, 10, 50).reshape(-1, 1) + with pytest.warns(DataWarning, match=r"\[-5\.0, 15\.0\] lie outside"): + transformer = BSplineTransformer(knot_locations=np.array([-5.0, 3.0, 3.0, 15.0])).fit(X) + np.testing.assert_array_equal(transformer.knots_[0][4:-4], [3.0]) + + +@pytest.mark.parametrize("knots", [[[1.0, 2.0]], ["a"], [np.nan]]) +def test_invalid_knot_locations_raise_a_typed_error(knots): + from pretab.exceptions import InvalidParamError + + with pytest.raises(InvalidParamError, match="knot_locations"): + BSplineTransformer(knot_locations=knots).fit(np.linspace(0, 10, 50).reshape(-1, 1)) diff --git a/tests/extension/test_extension_protocol.py b/tests/extension/test_extension_protocol.py index de89ae3..2ca252c 100644 --- a/tests/extension/test_extension_protocol.py +++ b/tests/extension/test_extension_protocol.py @@ -1,6 +1,7 @@ """Tests for the public ``BaseRepresentation`` extension base (P10.1).""" import numpy as np +import pandas as pd import pytest from sklearn.utils.validation import check_is_fitted @@ -62,6 +63,14 @@ def test_representation_spec_reflects_declaration(): assert spec.output_dim == 1 +def test_representation_spec_after_a_dataframe_fit_names_the_columns(): + X = pd.DataFrame({"length": np.linspace(0.0, 1.0, 20), "width": np.linspace(1.0, 2.0, 20)}) + est = _Square().fit(X) + spec = est.get_representation_spec() + assert spec.input_features == ("length", "width") + assert spec.output_features == tuple(est.get_feature_names_out()) + + def test_invalid_scope_rejected(): with pytest.raises(ValueError, match="scope"): diff --git a/tests/extension/test_registered_cross_fitting.py b/tests/extension/test_registered_cross_fitting.py new file mode 100644 index 0000000..3e8a039 --- /dev/null +++ b/tests/extension/test_registered_cross_fitting.py @@ -0,0 +1,69 @@ +"""Cross-fitting a Preprocessor that uses a registered, target-consuming class. + +Regression guard: a wrapped Preprocessor refits only its target-aware blocks per +fold, but a registered class without a ``RepresentationSpec`` (a plain +scikit-learn transformer such as ``TargetEncoder``) was never recognised as +target-aware. Its out-of-fold rows were then encoded by the all-data fit, which +had seen their own targets. +""" + +import warnings + +import numpy as np +import pandas as pd +import pytest +from sklearn.base import BaseEstimator, TransformerMixin +from sklearn.preprocessing import TargetEncoder + +from pretab import CrossFittedTransformer, Preprocessor, register_representation + + +class _MeanEncoder(TransformerMixin, BaseEstimator): + """Plain, optionally supervised encoder: category -> mean target when target-aware.""" + + def __init__(self, target_aware=True): + self.target_aware = target_aware + + def fit(self, X, y=None): + x = np.asarray(X, dtype=object)[:, 0] + if self.target_aware: + y = np.asarray(y, dtype=float) + self.mapping_ = {c: y[x == c].mean() for c in np.unique(x)} + else: + self.mapping_ = {c: float(i) for i, c in enumerate(np.unique(x))} + self.n_features_in_ = 1 + return self + + def transform(self, X): + return np.array([[self.mapping_.get(c, 0.0)] for c in np.asarray(X, dtype=object)[:, 0]]) + + def get_feature_names_out(self, input_features=None): + return np.asarray(input_features if input_features is not None else ["x0"], dtype=object) + + +@pytest.fixture +def noise_target_data(): + """An id column with four rows per id and a target of pure noise.""" + rng = np.random.default_rng(0) + X = pd.DataFrame({"x": rng.normal(size=400), "id": np.repeat([f"id{i}" for i in range(100)], 4)}) + return X, rng.normal(size=400) + + +@pytest.mark.parametrize( + ("cls", "supervision"), + [(TargetEncoder, "supervised"), (_MeanEncoder, "optional")], +) +def test_registered_target_encoder_is_refit_per_fold(noise_target_data, cls, supervision): + X, y = noise_target_data + register_representation("registered_encoder", cls, feature_kind="categorical", supervision=supervision) + pre = Preprocessor(numerical_method="minmax", categorical_method="registered_encoder", random_state=0) + + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + out_of_fold = np.asarray(CrossFittedTransformer(pre, n_folds=5, random_state=0).fit_transform(X, y)) + lineage = {item.output_feature: item.uses_target for item in pre.fit(X, y).get_feature_lineage()} + + # An encoding fit without each row's own target cannot predict pure noise; the + # all-data fit correlated at about 0.5. + assert abs(np.corrcoef(out_of_fold[:, -1], y)[0, 1]) < 0.2 + assert lineage["cat_id"] is True diff --git a/tests/integration/test_missing_policy.py b/tests/integration/test_missing_policy.py index ac728a3..483d9c3 100644 --- a/tests/integration/test_missing_policy.py +++ b/tests/integration/test_missing_policy.py @@ -316,3 +316,25 @@ def test_imputer_indicator_only_marks_features_with_missing_values_at_fit(clean_ pre = Preprocessor(numerical_method="minmax", add_missing_indicator=True).fit(clean_frame, y) assert not any("missing" in name for name in pre.get_feature_names_out()) assert np.asarray(pre.transform(frame_with_nan, return_array=True)).shape == (6, 2) + + +# --- representation summary of missing-indicator / separate-state blocks -------- + + +@pytest.mark.parametrize( + "options", + [ + {"add_missing_indicator": True}, + {"missing_policy": "impute_with_indicator"}, + {"missing_policy": "separate_state"}, + ], +) +def test_missing_state_blocks_report_their_representation(frame_with_nan, y, options): + """The spec summary read each block's last step, which for these blocks is the + FeatureUnion of the representation and the missing indicator, so to_spec() and + reproducibility_report() listed no representation for them.""" + plain = _bspline(missing_policy="impute").fit(frame_with_nan, y) + pre = _bspline(**options).fit(frame_with_nan, y) + + assert pre.reproducibility_report()["representations"] == {"a": "bspline", "b": "bspline"} + assert pre.to_spec()["representations"] == plain.to_spec()["representations"] diff --git a/tests/integration/test_output_format.py b/tests/integration/test_output_format.py index b16ccd3..f613dbc 100644 --- a/tests/integration/test_output_format.py +++ b/tests/integration/test_output_format.py @@ -276,3 +276,50 @@ def test_auto_format_is_fixed_at_fit_for_sparse_training_output(): def test_explicit_formats_are_stored_as_resolved(frame, y, output_format): pre = Preprocessor(numerical_method="minmax", output_format=output_format).fit(frame, y) assert pre.output_format_ == output_format + + +# --- global scikit-learn output config ---------------------------------------------- + + +@pytest.fixture +def mixed_frame(): + rng = np.random.default_rng(0) + a = rng.normal(size=40) + a[:4] = np.nan + return pd.DataFrame({"a": a, "color": rng.choice(["red", "green", "blue"], 40)}), rng.normal(size=40) + + +@pytest.mark.parametrize("options", [{"categorical_method": "one-hot"}, {"preset": "expanded"}]) +def test_one_hot_works_under_a_global_pandas_output_config(mixed_frame, options): + """The global transform_output reached the inner sparse OneHotEncoder, which refused it.""" + import sklearn + + X, y = mixed_frame + with sklearn.config_context(transform_output="pandas"): + pre = Preprocessor(**options).fit(X, y) + out = pre.transform(X) + assert isinstance(out, pd.DataFrame) + assert list(out.columns) == list(pre.get_feature_names_out()) + assert set(map(str, out.dtypes)) == {"float64"} + + +def test_missing_indicator_stays_numeric_under_a_global_pandas_output_config(mixed_frame): + import sklearn + + X, y = mixed_frame + with sklearn.config_context(transform_output="pandas"): + out = Preprocessor(numerical_method="minmax", add_missing_indicator=True).fit(X, y).transform(X) + assert isinstance(out, pd.DataFrame) + assert all(dtype.kind == "f" for dtype in out.dtypes) + np.testing.assert_array_equal(out["num_a__missingindicator_a"], X["a"].isna().astype(float)) + + +def test_global_polars_output_config_is_honoured(mixed_frame): + import sklearn + + pl = pytest.importorskip("polars") + X, y = mixed_frame + with sklearn.config_context(transform_output="polars"): + out = Preprocessor(categorical_method="one-hot").fit(X, y).transform(X) + assert isinstance(out, pl.DataFrame) + assert np.asarray(out).shape[0] == len(X) diff --git a/tests/integration/test_preprocessor.py b/tests/integration/test_preprocessor.py index 62148ef..04ba234 100644 --- a/tests/integration/test_preprocessor.py +++ b/tests/integration/test_preprocessor.py @@ -539,3 +539,59 @@ def test_pipeline_feature_names_propagate_through_an_array_step(named_numeric): names = pipe.get_feature_names_out() assert names is not None assert names.tolist() == ["num_age", "num_income"] + + +# --- positional input after a named fit (and vice versa) ---------------------------- + + +@pytest.fixture +def mixed_named_frame(): + rng = np.random.default_rng(0) + frame = pd.DataFrame( + { + "age": rng.normal(40, 10, 100), + "income": rng.normal(50, 10, 100), + "city": rng.choice(["a", "b", "c"], 100), + "flag": rng.random(100) > 0.5, + } + ) + return frame, rng.normal(size=100) + + +def test_frame_fit_accepts_an_array_of_the_fitted_width(mixed_named_frame): + """A DataFrame-fitted preprocessor rejected the same data as an array + ('columns are missing'), breaking NumPy-based serving and explainers.""" + from sklearn.linear_model import Ridge + from sklearn.pipeline import make_pipeline + + X, y = mixed_named_frame + pipe = make_pipeline(Preprocessor(random_state=0), Ridge()).fit(X, y) + with pytest.warns(UserWarning, match="does not have valid feature names"): + from_array = pipe.predict(X.to_numpy()) + np.testing.assert_allclose(from_array, pipe.predict(X)) + + +def test_array_fit_accepts_a_frame_of_the_fitted_width(mixed_named_frame): + X, y = mixed_named_frame + numeric = X[["age", "income"]] + pre = Preprocessor(numerical_method="minmax").fit(numeric.to_numpy(), y) + with pytest.warns(UserWarning, match="fitted without feature names"): + from_frame = pre.transform(numeric.rename(columns={"age": "u", "income": "v"})) + np.testing.assert_array_equal(from_frame, pre.transform(numeric.to_numpy())) + + +def test_integer_labelled_fit_matches_an_array_by_position(): + pre = Preprocessor(numerical_method="none", scaling="none").fit(pd.DataFrame({1: [10.0, 20, 30], 0: [1.0, 2, 3]})) + out = pre.transform(np.array([[10.0, 1.0], [30.0, 3.0]])) + np.testing.assert_array_equal(out, [[10.0, 1.0], [30.0, 3.0]]) + + +def test_positional_input_of_the_wrong_width_is_rejected(mixed_named_frame): + X, y = mixed_named_frame + pre = Preprocessor(random_state=0).fit(X, y) + with pytest.raises(PretabDataError, match="X has 3 features, but Preprocessor is expecting 4"): + pre.transform(X.to_numpy()[:, :3]) + numeric = X[["age", "income"]] + array_fit = Preprocessor(numerical_method="minmax").fit(numeric.to_numpy(), y) + with pytest.raises(PretabDataError, match="X has 1 features"): + array_fit.transform(X[["age"]]) diff --git a/tests/integration/test_serialization.py b/tests/integration/test_serialization.py index 77530fb..4cf5385 100644 --- a/tests/integration/test_serialization.py +++ b/tests/integration/test_serialization.py @@ -7,13 +7,18 @@ """ import json +import math +from typing import cast import numpy as np import pandas as pd import pytest +from sklearn.impute import SimpleImputer +from sklearn.pipeline import Pipeline from pretab import Preprocessor, PretabSerializationError, RepresentationPolicy from pretab.compose.serialize import SCHEMA_VERSION +from pretab.transformers import RBFExpansionTransformer @pytest.fixture @@ -105,6 +110,51 @@ def test_representation_summary_present(frame, target): assert "rbf" in families +@pytest.mark.parametrize("leaf_input", ["imputed", "frame"]) +def test_representation_summary_names_each_block_after_its_column(frame, target, leaf_input): + """Without an imputer or scaler each representation is fitted on its DataFrame + column; its spec raised there, and the summary silently came back empty. The + entries are named after the block's column instead of x0.""" + params = {"numerical_method": "rbf", "target_aware": False, "placement_strategy": "quantile"} + if leaf_input == "frame": + params.update(numerical_imputation=None, scaling=None) + p = Preprocessor(**params).fit(frame, target) + entries = {entry["columns"][0]: entry for entry in p.to_spec()["representations"]} + inputs = {column: entry["input_features"] for column, entry in entries.items()} + + assert p.reproducibility_report()["representations"] == {"a": "rbf", "b": "rbf", "c": "ordinal"} + assert inputs == {"a": ["a"], "b": ["b"], "c": ["c"]} + assert entries["a"]["output_features"] == [f"a_rbf{i}" for i in range(entries["a"]["output_dim"])] + + +def test_representation_summary_of_a_leaf_behind_the_imputer_indicator(frame, target): + """Before the #62 fix, ``add_missing_indicator`` put the imputer's built-in + indicator in front of the representation, so a preprocessor fitted then has a + leaf with two inputs for a one-column block. Its summary entry names its own + inputs instead of failing on the single column name.""" + X = frame.copy() + X.loc[::7, "a"] = np.nan + p = Preprocessor(numerical_method="rbf", target_aware=False, placement_strategy="quantile").fit(X, target) + legacy_block = Pipeline( + [ + ("imputer", SimpleImputer(strategy="median", add_indicator=True)), + ("rbf", RBFExpansionTransformer(target_aware=False, placement_strategy="quantile")), + ] + ).fit(X[["a"]], target) + column_transformer = p.column_transformer_ + column_transformer.transformers_ = [ + (name, legacy_block if columns == ["a"] else transformer, columns) + for name, transformer, columns in column_transformer.transformers_ + ] + + entries = {entry["columns"][0]: entry for entry in p.to_spec()["representations"]} + + assert entries["a"]["input_features"] == ["x0", "x1"] + assert entries["b"]["input_features"] == ["b"] + assert p.reproducibility_report()["representations"] == {"a": "rbf", "b": "rbf", "c": "ordinal"} + assert isinstance(p.fingerprint_, str) + + def test_round_trip_preserves_dtype_and_output_format(frame, target): p = Preprocessor( numerical_method="bspline", @@ -338,3 +388,62 @@ def test_object_arrays_written_before_element_encoding_still_load(): assert isinstance(decoded, np.ndarray) assert decoded.dtype == object assert decoded.tolist() == [["a", None], [True, 1.5]] + + +# --- strict JSON: no NaN / Infinity tokens -------------------------------------------- + + +def _strict_loads(text): + def reject(token): + raise ValueError(f"non-standard JSON token {token}") + + return json.loads(text, parse_constant=reject) + + +@pytest.fixture +def frame_with_missing(): + rng = np.random.default_rng(0) + a = rng.normal(size=80) + a[:6] = np.nan + X = pd.DataFrame({"a": a, "b": rng.exponential(size=80), "c": rng.choice(["x", "y"], 80)}) + return X, rng.normal(size=80) + + +def test_default_spec_file_is_strict_json(frame_with_missing, tmp_path): + """The imputers' NaN missing-value marker was written as a bare NaN token, + which strict JSON parsers (JavaScript, Go, Rust, ...) reject.""" + X, y = frame_with_missing + pre = Preprocessor(random_state=0).fit(X, y) + path = tmp_path / "spec.json" + pre.to_spec(path) + text = path.read_text(encoding="utf-8") + assert "NaN" not in text and "Infinity" not in text + loaded = Preprocessor.from_spec(_strict_loads(text)) + np.testing.assert_array_equal(np.asarray(loaded.transform(X)), np.asarray(pre.transform(X))) + assert loaded.fingerprint_ == pre.fingerprint_ + + +def test_non_finite_floats_round_trip(): + from pretab.compose.serialize import _decode, _encode + + array = np.array([[1.0, np.nan], [np.inf, -np.inf]]) + decoded = _decode(_strict_loads(json.dumps(_encode(array), allow_nan=False))) + assert isinstance(decoded, np.ndarray) + np.testing.assert_array_equal(decoded, array) + assert decoded.dtype == array.dtype + nan, neg_inf = ( + float(cast(float, _decode(_strict_loads(json.dumps(_encode(value), allow_nan=False))))) + for value in (np.nan, -np.inf) + ) + assert math.isnan(nan) and neg_inf == -math.inf + + +def test_specs_with_bare_nan_tokens_still_load(frame_with_missing): + X, y = frame_with_missing + pre = Preprocessor(random_state=0).fit(X, y) + spec = json.loads(json.dumps(pre.to_spec())) + # Rewrite the tagged floats the way older versions wrote them: as bare NaN. + legacy_text = json.dumps(spec).replace('{"__float__": "nan"}', "NaN") + assert "NaN" in legacy_text + loaded = Preprocessor.from_spec(json.loads(legacy_text)) + np.testing.assert_array_equal(np.asarray(loaded.transform(X)), np.asarray(pre.transform(X))) diff --git a/tests/integration/test_verbosity.py b/tests/integration/test_verbosity.py index 8b44f78..3797978 100644 --- a/tests/integration/test_verbosity.py +++ b/tests/integration/test_verbosity.py @@ -120,6 +120,17 @@ def test_verbose_3_logs_internal_decisions(sample_data, caplog): assert "thresholds_" in debug_text or "total_output_dim_" in debug_text +@pytest.mark.parametrize("options", [{"add_missing_indicator": True}, {"missing_policy": "separate_state"}]) +def test_verbose_3_logs_internal_decisions_of_missing_state_blocks(sample_data, caplog, options): + """These blocks are a FeatureUnion of the representation and the missing + indicator; the union itself has no fitted thresholds, so none were logged.""" + X, y = sample_data + caplog.set_level(logging.DEBUG, logger="pretab") + Preprocessor(numerical_method="ple", verbose=3, **options).fit(X, y) + debug_text = "\n".join(r.getMessage() for r in caplog.records if r.levelno == logging.DEBUG) + assert "num_num1.thresholds_" in debug_text + + def test_verbose_true_behaves_like_level_1(sample_data, caplog): X, y = sample_data caplog.set_level(logging.DEBUG, logger="pretab") diff --git a/tests/placement/test_placement.py b/tests/placement/test_placement.py index d0dc463..58edd88 100644 --- a/tests/placement/test_placement.py +++ b/tests/placement/test_placement.py @@ -215,3 +215,44 @@ def test_rbf_adapter_unsupervised_matches_inline(data): assert np.allclose(centers, np.percentile(x, np.linspace(0, 100, 6))) centers_u = RBFPlacementAdapter(target_aware=False, placement_strategy="uniform").get_centers(x, None, 6, 6) assert np.allclose(centers_u, np.linspace(x.min(), x.max(), 6)) + + +# --- quantile placement on tied data ------------------------------------------------ + + +def _tied_features(): + rng = np.random.default_rng(0) + return { + "zero-inflated": np.where(rng.random(1000) < 0.6, 0.0, rng.exponential(5.0, 1000)), + "discrete": rng.integers(0, 5, 1000).astype(float), + "top-coded": np.minimum(rng.uniform(0, 160, 1000), 100.0), + } + + +@pytest.mark.parametrize("name", ["zero-inflated", "discrete", "top-coded"]) +@pytest.mark.parametrize("include_endpoints", [True, False]) +def test_quantile_placement_returns_distinct_locations_on_tied_data(name, include_endpoints): + """Coinciding quantiles repeated feature-map centers, duplicating output columns.""" + x = _tied_features()[name] + locations = QuantilePlacement(8, include_endpoints=include_endpoints).fit(x).locations_ + assert len(locations) == 8 + assert len(np.unique(locations)) == 8 + if include_endpoints: + assert locations[0] == x.min() and locations[-1] == x.max() + else: + assert locations.min() > x.min() and locations.max() < x.max() + + +def test_quantile_placement_is_unchanged_on_untied_data(): + x = np.random.default_rng(1).normal(size=500) + np.testing.assert_array_equal( + QuantilePlacement(7, include_endpoints=True).fit(x).locations_, np.quantile(x, np.linspace(0, 1, 7)) + ) + + +def test_quantile_feature_map_has_no_duplicate_columns_on_tied_data(): + from pretab.transformers import RBFExpansionTransformer + + X = _tied_features()["zero-inflated"].reshape(-1, 1) + out = RBFExpansionTransformer(output_dim=8, placement_strategy="quantile").fit(X).transform(X) + assert len(np.unique(out.round(10), axis=1).T) == 8 diff --git a/tests/transformers/test_feature_names_in.py b/tests/transformers/test_feature_names_in.py new file mode 100644 index 0000000..a9f2428 --- /dev/null +++ b/tests/transformers/test_feature_names_in.py @@ -0,0 +1,112 @@ +"""Standalone numerical transformers follow scikit-learn's feature-name contract. + +Fitted on a DataFrame they record ``feature_names_in_``, name their outputs after +the columns, reject renamed or reordered columns at ``transform`` and validate +``get_feature_names_out(input_features)``. Before, column names were dropped +(outputs were named ``x0_...``) and a reordered frame was silently accepted. +""" + +import warnings + +import numpy as np +import pandas as pd +import pytest +from sklearn.utils.estimator_checks import ( + check_dataframe_column_names_consistency, + check_transformer_get_feature_names_out_pandas, +) + +from pretab.transformers import ( + BSplineTransformer, + CubicRegressionSplineTransformer, + FourierFeatureTransformer, + NumericBinningTransformer, + PLETransformer, + RBFExpansionTransformer, + ReLUExpansionTransformer, + TensorProductSplineTransformer, +) + +pytestmark = pytest.mark.filterwarnings("ignore::pretab.exceptions.LeakageWarning") + +TRANSFORMERS = [ + lambda: RBFExpansionTransformer(output_dim=3), + lambda: ReLUExpansionTransformer(output_dim=3), + lambda: PLETransformer(output_dim=3, random_state=0), + lambda: NumericBinningTransformer(output_dim=3, encode="onehot"), + lambda: BSplineTransformer(output_dim=5), + lambda: CubicRegressionSplineTransformer(), + lambda: FourierFeatureTransformer(), + lambda: TensorProductSplineTransformer(), +] +IDS = ["rbf", "relu", "ple", "binning", "bspline", "cubicspline", "fourier", "tensorspline"] + + +@pytest.fixture +def frame(): + rng = np.random.default_rng(0) + X = pd.DataFrame({"age": rng.normal(40, 10, 100), "income": rng.uniform(1, 5, 100)}) + return X, X["age"].to_numpy() / 10 + rng.normal(size=100) + + +@pytest.mark.parametrize("make", TRANSFORMERS, ids=IDS) +def test_dataframe_fit_records_feature_names_and_uses_them(make, frame): + X, y = frame + transformer = make().fit(X, y) + assert list(transformer.feature_names_in_) == ["age", "income"] + names = list(transformer.get_feature_names_out()) + assert all(("age" in name) or ("income" in name) for name in names) + assert not any(name.startswith("x0") for name in names) + + +@pytest.mark.parametrize("make", TRANSFORMERS, ids=IDS) +def test_reordered_columns_are_rejected(make, frame): + X, y = frame + transformer = make().fit(X, y) + with pytest.raises(ValueError, match="feature names"): + transformer.transform(X[["income", "age"]]) + + +@pytest.mark.parametrize("make", TRANSFORMERS, ids=IDS) +def test_array_after_a_dataframe_fit_warns_like_scikit_learn(make, frame): + X, y = frame + transformer = make().fit(X, y) + with pytest.warns(UserWarning, match="does not have valid feature names"): + transformer.transform(X.to_numpy()) + + +@pytest.mark.parametrize("make", TRANSFORMERS, ids=IDS) +def test_input_features_are_validated(make, frame): + X, y = frame + transformer = make().fit(X, y) + with pytest.raises(ValueError): + transformer.get_feature_names_out(["age"]) + with pytest.raises(ValueError, match="not equal to feature_names_in_"): + transformer.get_feature_names_out(["a", "b"]) + + +@pytest.mark.parametrize("make", TRANSFORMERS, ids=IDS) +def test_array_fit_keeps_generic_names_and_drops_stale_feature_names(make, frame): + X, y = frame + transformer = make().fit(X, y).fit(X.to_numpy(), y) + assert not hasattr(transformer, "feature_names_in_") + assert "x0" in str(transformer.get_feature_names_out()[0]) + + +@pytest.mark.parametrize( + "make", + [ + lambda: RBFExpansionTransformer(output_dim=3), + lambda: PLETransformer(output_dim=3, random_state=0), + lambda: NumericBinningTransformer(output_dim=3), + lambda: BSplineTransformer(output_dim=5), + ], + ids=["rbf", "ple", "binning", "bspline"], +) +def test_scikit_learn_feature_name_checks_pass(make): + estimator = make() + name = type(estimator).__name__ + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + check_dataframe_column_names_consistency(name, estimator) + check_transformer_get_feature_names_out_pandas(name, estimator)