From 8bf5fb9b6b95dfc2b841811f11c92ea4248ce5d2 Mon Sep 17 00:00:00 2001 From: stephantul Date: Sun, 27 Sep 2026 19:12:01 +0200 Subject: [PATCH 1/2] feat(train): allow pair models without a head A pair similarity model with n_layers=0 whose output dimension equals the embedding dimension now has an empty head, so the embeddings are trained and used as is, and to_static_model gives the same embeddings as the trained model. Classifiers and regressors always keep their linear layer. MLPHead and the ONNX pipeline wrapper now pass the input through unchanged when the head has no layers, instead of failing to unpack an empty list. --- model2vec/inference/mlp.py | 4 +++- model2vec/onnx.py | 11 +++++++---- model2vec/train/pairs.py | 9 ++++++++- tests/test_export_to_onnx.py | 19 ++++++++++++++++++- tests/test_trainable.py | 30 ++++++++++++++++++++++++++++++ 5 files changed, 66 insertions(+), 7 deletions(-) diff --git a/model2vec/inference/mlp.py b/model2vec/inference/mlp.py index a1bcd51..fe8e8ed 100644 --- a/model2vec/inference/mlp.py +++ b/model2vec/inference/mlp.py @@ -43,7 +43,7 @@ def __init__( ) -> None: """An MLP with ReLU activation. - :param layers: The linear layers, in order. + :param layers: The linear layers, in order. If empty, the input is passed through unchanged. :param activation: The output activation. :param classes: The classes, if the task is a classification task. """ @@ -54,6 +54,8 @@ def __init__( def _logits(self, X: np.ndarray) -> np.ndarray: """Run the forward through the layers.""" out = X + if not self.layers: + return out *hidden_layers, last_layer = self.layers for layer in hidden_layers: out = np.maximum(layer(out), 0.0) diff --git a/model2vec/onnx.py b/model2vec/onnx.py index 73900e7..e1f12aa 100644 --- a/model2vec/onnx.py +++ b/model2vec/onnx.py @@ -124,10 +124,13 @@ def __init__(self, pipeline: StaticModelPipeline) -> None: def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor: """Encode the inputs and run them through the head, applying the output activation.""" out = self.encoder(input_ids, attention_mask).float() - *hidden_layers, last_layer = self.layers - for layer in hidden_layers: - out = torch.relu(layer(out)) - logits = last_layer(out) + if len(self.layers) == 0: + logits = out + else: + *hidden_layers, last_layer = self.layers + for layer in hidden_layers: + out = torch.relu(layer(out)) + logits = last_layer(out) if self.activation == Activation.SOFTMAX: return torch.softmax(logits, dim=-1) if self.activation == Activation.SIGMOID: diff --git a/model2vec/train/pairs.py b/model2vec/train/pairs.py index 07f9254..6850f42 100644 --- a/model2vec/train/pairs.py +++ b/model2vec/train/pairs.py @@ -54,7 +54,8 @@ def __init__( :param vectors: The embeddings of the staticmodel. :param tokenizer: The tokenizer. - :param n_layers: The number of layers in the head. + :param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding + dimension, the model has no head, and the embeddings are used as is. :param hidden_dim: The hidden dimension of the head. :param out_dim: The output embedding dimension. If None, defaults to the input embedding dimension. :param pad_id: The padding id. This is set to 0 in almost all model2vec models. @@ -80,6 +81,12 @@ def __init__( max_length=max_length, ) + def construct_head(self) -> nn.Sequential: + """Construct the head, which is empty if it has no layers and doesn't change the dimension.""" + if self.n_layers == 0 and self.embed_dim == self.out_dim: + return nn.Sequential() + return super().construct_head() + def forward(self, input_ids: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: # type: ignore[override] """Encode both halves of a pair batch through the shared embeddings and head. diff --git a/tests/test_export_to_onnx.py b/tests/test_export_to_onnx.py index b229d24..04df5f8 100644 --- a/tests/test_export_to_onnx.py +++ b/tests/test_export_to_onnx.py @@ -27,7 +27,7 @@ _save_tokenizer_and_config, export_model_to_onnx, ) -from model2vec.train import StaticModelForClassification +from model2vec.train import StaticModelForClassification, StaticModelForPairSimilarity def _tokenize(pipeline: StaticModelPipeline, texts: list[str]) -> tuple[torch.Tensor, torch.Tensor]: @@ -147,6 +147,23 @@ def test_pipeline_onnx_matches_projector( np.testing.assert_allclose(onnx_output, expected, atol=1e-4) +def test_pipeline_onnx_matches_empty_head(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer, tmp_path: Path) -> None: + """A pipeline whose head has no layers exports the static model's embeddings.""" + model = StaticModelForPairSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + pipeline = model.to_pipeline() + assert pipeline.head.layers == [] + texts = ["dog", "cat"] + torch_model = TorchStaticModelPipeline(pipeline) + input_ids, attention_mask = _tokenize(pipeline, texts) + + onnx_output = _export(torch_model, input_ids, attention_mask, tmp_path / "model.onnx") + expected = pipeline.predict(texts, use_multiprocessing=False) + + np.testing.assert_allclose(onnx_output, expected, atol=1e-4) + + def test_save_tokenizer_and_config_removes_post_processor_by_default( mock_static_model: StaticModel, tmp_path: Path ) -> None: diff --git a/tests/test_trainable.py b/tests/test_trainable.py index 7d90fb0..0910ea0 100644 --- a/tests/test_trainable.py +++ b/tests/test_trainable.py @@ -445,6 +445,36 @@ def test_pair_similarity_out_dim_defaults_to_embed_dim(mock_vectors: np.ndarray, assert s.out_dim == 7 +def test_pair_similarity_no_head(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """Without layers and with an unchanged dimension, the model has no head and matches its static model.""" + model = StaticModelForPairSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + assert len(model.head) == 0 + + texts = ["dog cat", "dog"] + np.testing.assert_allclose(model.encode(texts), model.to_static_model().encode(texts), atol=1e-6) + np.testing.assert_allclose(model.encode(texts), model.to_pipeline().predict(texts), atol=1e-6) + + +def test_pair_similarity_head_changes_dimension(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """Without layers but with a different output dimension, the head is a single linear layer.""" + model = StaticModelForPairSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0, out_dim=3 + ) + assert len(model.head) == 1 + assert isinstance(model.head[0], torch.nn.Linear) + + +def test_classifier_keeps_head_when_dimensions_match(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """A classifier without layers keeps its linear layer, even if the number of classes equals the dimension.""" + model = StaticModelForClassification( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + assert model.out_dim == mock_vectors.shape[1] + assert isinstance(model.head[0], torch.nn.Linear) + + def test_pair_similarity_forward(mock_trained_pair_similarity_pipeline: StaticModelForPairSimilarity) -> None: """The forward pass should return one head output per half of the pair batch.""" model = mock_trained_pair_similarity_pipeline From 019f2f1a1a560194bac6661617581cf506a1b9ad Mon Sep 17 00:00:00 2001 From: stephantul Date: Sun, 27 Sep 2026 20:03:30 +0200 Subject: [PATCH 2/2] make similarity not have a head if n_layers == 0 --- model2vec/train/base.py | 7 ++++-- model2vec/train/classifier.py | 9 ++++++++ model2vec/train/pairs.py | 6 ----- tests/test_trainable.py | 42 +++++++++++++++++++++++++++++------ 4 files changed, 49 insertions(+), 15 deletions(-) diff --git a/model2vec/train/base.py b/model2vec/train/base.py index 982176e..ef2317a 100644 --- a/model2vec/train/base.py +++ b/model2vec/train/base.py @@ -50,7 +50,8 @@ def __init__( :param vectors: The embeddings of the staticmodel. :param tokenizer: The tokenizer. :param hidden_dim: The hidden dimension of the head. - :param n_layers: The number of layers in the head. + :param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding + dimension, the model has no head and the embeddings are used as is. :param out_dim: The output dimension of the head. :param pad_id: The padding id. This is set to 0 in almost all model2vec models :param token_mapping: The token mapping. If None, the token mapping is set to the range of the number of vectors. @@ -111,7 +112,9 @@ def construct_weights(self) -> nn.Parameter: return nn.Parameter(w, requires_grad=not self.freeze_weights) def construct_head(self) -> nn.Sequential: - """Constructs a simple classifier head.""" + """Constructs a simple head, which is empty if it has no layers and doesn't change the dimension.""" + if self.n_layers == 0 and self.embed_dim == self.out_dim: + return nn.Sequential() modules: list[nn.Module] = [] if self.n_layers == 0: modules.append(nn.Linear(self.embed_dim, self.out_dim)) diff --git a/model2vec/train/classifier.py b/model2vec/train/classifier.py index cb23568..93c3cd5 100644 --- a/model2vec/train/classifier.py +++ b/model2vec/train/classifier.py @@ -87,6 +87,15 @@ def classes(self) -> np.ndarray: """Return all clasess in the correct order.""" return np.array(self.classes_) + def construct_head(self) -> nn.Sequential: + """Constructs a classifier head, which always has at least one linear layer.""" + if self.n_layers == 0: + linear = nn.Linear(self.embed_dim, self.out_dim) + nn.init.xavier_uniform_(linear.weight) + nn.init.zeros_(linear.bias) + return nn.Sequential(linear) + return super().construct_head() + def predict( self, X: list[str], show_progress_bar: bool = False, batch_size: int = 1024, threshold: float = 0.5 ) -> np.ndarray: diff --git a/model2vec/train/pairs.py b/model2vec/train/pairs.py index 6850f42..672bf84 100644 --- a/model2vec/train/pairs.py +++ b/model2vec/train/pairs.py @@ -81,12 +81,6 @@ def __init__( max_length=max_length, ) - def construct_head(self) -> nn.Sequential: - """Construct the head, which is empty if it has no layers and doesn't change the dimension.""" - if self.n_layers == 0 and self.embed_dim == self.out_dim: - return nn.Sequential() - return super().construct_head() - def forward(self, input_ids: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: # type: ignore[override] """Encode both halves of a pair batch through the shared embeddings and head. diff --git a/tests/test_trainable.py b/tests/test_trainable.py index 0910ea0..034254f 100644 --- a/tests/test_trainable.py +++ b/tests/test_trainable.py @@ -1,5 +1,6 @@ import logging from tempfile import TemporaryDirectory +from typing import Any import numpy as np import pytest @@ -44,7 +45,7 @@ def test_init_base_class(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> """Test successful initialization of the base class.""" vectors_torched = torch.from_numpy(mock_vectors) s = BaseFinetuneable( - vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=2, n_layers=0, pad_id=0 + vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=3, n_layers=0, pad_id=0 ) assert s.vectors.shape == mock_vectors.shape assert s.w.shape[0] == mock_vectors.shape[0] @@ -445,10 +446,16 @@ def test_pair_similarity_out_dim_defaults_to_embed_dim(mock_vectors: np.ndarray, assert s.out_dim == 7 -def test_pair_similarity_no_head(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: +@pytest.mark.parametrize( + "model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity] +) +def test_no_head_without_layers(model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: """Without layers and with an unchanged dimension, the model has no head and matches its static model.""" - model = StaticModelForPairSimilarity( - vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + model = model_class( + vectors=torch.from_numpy(mock_vectors).float(), + tokenizer=mock_tokenizer, + n_layers=0, + out_dim=mock_vectors.shape[1], ) assert len(model.head) == 0 @@ -457,15 +464,36 @@ def test_pair_similarity_no_head(mock_vectors: np.ndarray, mock_tokenizer: Token np.testing.assert_allclose(model.encode(texts), model.to_pipeline().predict(texts), atol=1e-6) -def test_pair_similarity_head_changes_dimension(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: +@pytest.mark.parametrize( + "model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity] +) +def test_head_without_layers_changes_dimension( + model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer +) -> None: """Without layers but with a different output dimension, the head is a single linear layer.""" - model = StaticModelForPairSimilarity( - vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0, out_dim=3 + model = model_class( + vectors=torch.from_numpy(mock_vectors).float(), + tokenizer=mock_tokenizer, + n_layers=0, + out_dim=mock_vectors.shape[1] + 1, ) assert len(model.head) == 1 assert isinstance(model.head[0], torch.nn.Linear) +def test_similarity_fit_without_layers(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """The head of a similarity model follows the dimension of the targets it is fit on.""" + model = StaticModelForSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + texts = ["word1", "word2", "word3", "word1 word2"] * 2 + model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1]), max_epochs=1) + assert len(model.head) == 0 + + model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1] + 1), max_epochs=1) + assert isinstance(model.head[0], torch.nn.Linear) + + def test_classifier_keeps_head_when_dimensions_match(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: """A classifier without layers keeps its linear layer, even if the number of classes equals the dimension.""" model = StaticModelForClassification(