diff --git a/model2vec/inference/mlp.py b/model2vec/inference/mlp.py index a1bcd51..fe8e8ed 100644 --- a/model2vec/inference/mlp.py +++ b/model2vec/inference/mlp.py @@ -43,7 +43,7 @@ def __init__( ) -> None: """An MLP with ReLU activation. - :param layers: The linear layers, in order. + :param layers: The linear layers, in order. If empty, the input is passed through unchanged. :param activation: The output activation. :param classes: The classes, if the task is a classification task. """ @@ -54,6 +54,8 @@ def __init__( def _logits(self, X: np.ndarray) -> np.ndarray: """Run the forward through the layers.""" out = X + if not self.layers: + return out *hidden_layers, last_layer = self.layers for layer in hidden_layers: out = np.maximum(layer(out), 0.0) diff --git a/model2vec/onnx.py b/model2vec/onnx.py index 73900e7..e1f12aa 100644 --- a/model2vec/onnx.py +++ b/model2vec/onnx.py @@ -124,10 +124,13 @@ def __init__(self, pipeline: StaticModelPipeline) -> None: def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor: """Encode the inputs and run them through the head, applying the output activation.""" out = self.encoder(input_ids, attention_mask).float() - *hidden_layers, last_layer = self.layers - for layer in hidden_layers: - out = torch.relu(layer(out)) - logits = last_layer(out) + if len(self.layers) == 0: + logits = out + else: + *hidden_layers, last_layer = self.layers + for layer in hidden_layers: + out = torch.relu(layer(out)) + logits = last_layer(out) if self.activation == Activation.SOFTMAX: return torch.softmax(logits, dim=-1) if self.activation == Activation.SIGMOID: diff --git a/model2vec/train/base.py b/model2vec/train/base.py index 982176e..ef2317a 100644 --- a/model2vec/train/base.py +++ b/model2vec/train/base.py @@ -50,7 +50,8 @@ def __init__( :param vectors: The embeddings of the staticmodel. :param tokenizer: The tokenizer. :param hidden_dim: The hidden dimension of the head. - :param n_layers: The number of layers in the head. + :param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding + dimension, the model has no head and the embeddings are used as is. :param out_dim: The output dimension of the head. :param pad_id: The padding id. This is set to 0 in almost all model2vec models :param token_mapping: The token mapping. If None, the token mapping is set to the range of the number of vectors. @@ -111,7 +112,9 @@ def construct_weights(self) -> nn.Parameter: return nn.Parameter(w, requires_grad=not self.freeze_weights) def construct_head(self) -> nn.Sequential: - """Constructs a simple classifier head.""" + """Constructs a simple head, which is empty if it has no layers and doesn't change the dimension.""" + if self.n_layers == 0 and self.embed_dim == self.out_dim: + return nn.Sequential() modules: list[nn.Module] = [] if self.n_layers == 0: modules.append(nn.Linear(self.embed_dim, self.out_dim)) diff --git a/model2vec/train/classifier.py b/model2vec/train/classifier.py index cb23568..93c3cd5 100644 --- a/model2vec/train/classifier.py +++ b/model2vec/train/classifier.py @@ -87,6 +87,15 @@ def classes(self) -> np.ndarray: """Return all clasess in the correct order.""" return np.array(self.classes_) + def construct_head(self) -> nn.Sequential: + """Constructs a classifier head, which always has at least one linear layer.""" + if self.n_layers == 0: + linear = nn.Linear(self.embed_dim, self.out_dim) + nn.init.xavier_uniform_(linear.weight) + nn.init.zeros_(linear.bias) + return nn.Sequential(linear) + return super().construct_head() + def predict( self, X: list[str], show_progress_bar: bool = False, batch_size: int = 1024, threshold: float = 0.5 ) -> np.ndarray: diff --git a/model2vec/train/pairs.py b/model2vec/train/pairs.py index 07f9254..672bf84 100644 --- a/model2vec/train/pairs.py +++ b/model2vec/train/pairs.py @@ -54,7 +54,8 @@ def __init__( :param vectors: The embeddings of the staticmodel. :param tokenizer: The tokenizer. - :param n_layers: The number of layers in the head. + :param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding + dimension, the model has no head, and the embeddings are used as is. :param hidden_dim: The hidden dimension of the head. :param out_dim: The output embedding dimension. If None, defaults to the input embedding dimension. :param pad_id: The padding id. This is set to 0 in almost all model2vec models. diff --git a/tests/test_export_to_onnx.py b/tests/test_export_to_onnx.py index b229d24..04df5f8 100644 --- a/tests/test_export_to_onnx.py +++ b/tests/test_export_to_onnx.py @@ -27,7 +27,7 @@ _save_tokenizer_and_config, export_model_to_onnx, ) -from model2vec.train import StaticModelForClassification +from model2vec.train import StaticModelForClassification, StaticModelForPairSimilarity def _tokenize(pipeline: StaticModelPipeline, texts: list[str]) -> tuple[torch.Tensor, torch.Tensor]: @@ -147,6 +147,23 @@ def test_pipeline_onnx_matches_projector( np.testing.assert_allclose(onnx_output, expected, atol=1e-4) +def test_pipeline_onnx_matches_empty_head(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer, tmp_path: Path) -> None: + """A pipeline whose head has no layers exports the static model's embeddings.""" + model = StaticModelForPairSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + pipeline = model.to_pipeline() + assert pipeline.head.layers == [] + texts = ["dog", "cat"] + torch_model = TorchStaticModelPipeline(pipeline) + input_ids, attention_mask = _tokenize(pipeline, texts) + + onnx_output = _export(torch_model, input_ids, attention_mask, tmp_path / "model.onnx") + expected = pipeline.predict(texts, use_multiprocessing=False) + + np.testing.assert_allclose(onnx_output, expected, atol=1e-4) + + def test_save_tokenizer_and_config_removes_post_processor_by_default( mock_static_model: StaticModel, tmp_path: Path ) -> None: diff --git a/tests/test_trainable.py b/tests/test_trainable.py index 7d90fb0..034254f 100644 --- a/tests/test_trainable.py +++ b/tests/test_trainable.py @@ -1,5 +1,6 @@ import logging from tempfile import TemporaryDirectory +from typing import Any import numpy as np import pytest @@ -44,7 +45,7 @@ def test_init_base_class(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> """Test successful initialization of the base class.""" vectors_torched = torch.from_numpy(mock_vectors) s = BaseFinetuneable( - vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=2, n_layers=0, pad_id=0 + vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=3, n_layers=0, pad_id=0 ) assert s.vectors.shape == mock_vectors.shape assert s.w.shape[0] == mock_vectors.shape[0] @@ -445,6 +446,63 @@ def test_pair_similarity_out_dim_defaults_to_embed_dim(mock_vectors: np.ndarray, assert s.out_dim == 7 +@pytest.mark.parametrize( + "model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity] +) +def test_no_head_without_layers(model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """Without layers and with an unchanged dimension, the model has no head and matches its static model.""" + model = model_class( + vectors=torch.from_numpy(mock_vectors).float(), + tokenizer=mock_tokenizer, + n_layers=0, + out_dim=mock_vectors.shape[1], + ) + assert len(model.head) == 0 + + texts = ["dog cat", "dog"] + np.testing.assert_allclose(model.encode(texts), model.to_static_model().encode(texts), atol=1e-6) + np.testing.assert_allclose(model.encode(texts), model.to_pipeline().predict(texts), atol=1e-6) + + +@pytest.mark.parametrize( + "model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity] +) +def test_head_without_layers_changes_dimension( + model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer +) -> None: + """Without layers but with a different output dimension, the head is a single linear layer.""" + model = model_class( + vectors=torch.from_numpy(mock_vectors).float(), + tokenizer=mock_tokenizer, + n_layers=0, + out_dim=mock_vectors.shape[1] + 1, + ) + assert len(model.head) == 1 + assert isinstance(model.head[0], torch.nn.Linear) + + +def test_similarity_fit_without_layers(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """The head of a similarity model follows the dimension of the targets it is fit on.""" + model = StaticModelForSimilarity( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + texts = ["word1", "word2", "word3", "word1 word2"] * 2 + model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1]), max_epochs=1) + assert len(model.head) == 0 + + model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1] + 1), max_epochs=1) + assert isinstance(model.head[0], torch.nn.Linear) + + +def test_classifier_keeps_head_when_dimensions_match(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None: + """A classifier without layers keeps its linear layer, even if the number of classes equals the dimension.""" + model = StaticModelForClassification( + vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0 + ) + assert model.out_dim == mock_vectors.shape[1] + assert isinstance(model.head[0], torch.nn.Linear) + + def test_pair_similarity_forward(mock_trained_pair_similarity_pipeline: StaticModelForPairSimilarity) -> None: """The forward pass should return one head output per half of the pair batch.""" model = mock_trained_pair_similarity_pipeline