Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion model2vec/inference/mlp.py
Original file line number Diff line number Diff line change
Expand Up @@ -43,7 +43,7 @@ def __init__(
) -> None:
"""An MLP with ReLU activation.

:param layers: The linear layers, in order.
:param layers: The linear layers, in order. If empty, the input is passed through unchanged.
:param activation: The output activation.
:param classes: The classes, if the task is a classification task.
"""
Expand All @@ -54,6 +54,8 @@ def __init__(
def _logits(self, X: np.ndarray) -> np.ndarray:
"""Run the forward through the layers."""
out = X
if not self.layers:
return out
*hidden_layers, last_layer = self.layers
for layer in hidden_layers:
out = np.maximum(layer(out), 0.0)
Expand Down
11 changes: 7 additions & 4 deletions model2vec/onnx.py
Original file line number Diff line number Diff line change
Expand Up @@ -124,10 +124,13 @@ def __init__(self, pipeline: StaticModelPipeline) -> None:
def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor:
"""Encode the inputs and run them through the head, applying the output activation."""
out = self.encoder(input_ids, attention_mask).float()
*hidden_layers, last_layer = self.layers
for layer in hidden_layers:
out = torch.relu(layer(out))
logits = last_layer(out)
if len(self.layers) == 0:
logits = out
else:
*hidden_layers, last_layer = self.layers
for layer in hidden_layers:
out = torch.relu(layer(out))
logits = last_layer(out)
if self.activation == Activation.SOFTMAX:
return torch.softmax(logits, dim=-1)
if self.activation == Activation.SIGMOID:
Expand Down
7 changes: 5 additions & 2 deletions model2vec/train/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,7 +50,8 @@ def __init__(
:param vectors: The embeddings of the staticmodel.
:param tokenizer: The tokenizer.
:param hidden_dim: The hidden dimension of the head.
:param n_layers: The number of layers in the head.
:param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding
dimension, the model has no head and the embeddings are used as is.
:param out_dim: The output dimension of the head.
:param pad_id: The padding id. This is set to 0 in almost all model2vec models
:param token_mapping: The token mapping. If None, the token mapping is set to the range of the number of vectors.
Expand Down Expand Up @@ -111,7 +112,9 @@ def construct_weights(self) -> nn.Parameter:
return nn.Parameter(w, requires_grad=not self.freeze_weights)

def construct_head(self) -> nn.Sequential:
"""Constructs a simple classifier head."""
"""Constructs a simple head, which is empty if it has no layers and doesn't change the dimension."""
if self.n_layers == 0 and self.embed_dim == self.out_dim:
return nn.Sequential()
Comment on lines +116 to +117

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P1 Headless training loses gradient path When n_layers=0, the dimensions match, and both freeze=True and freeze_weights=True, this empty head leaves nothing trainable. The training loop still calls loss.backward(), but the loss has no gradient path, so fitting fails. Previously, the linear head allowed this head-only training configuration to work.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Hmm user error? They would need to set all of these options explicitly.

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That’s fair for the new headless configuration: explicitly setting both freeze=True and freeze_weights=True means the caller has asked for every parameter to be frozen, so there is no meaningful optimization to perform. I’d still avoid letting this fail later at loss.backward() with an opaque “does not require grad” error, because this combination worked before when the linear head remained trainable. A small validation in the training path (or at initialization) that raises a clear error when the model has no trainable parameters would make the behavior intentional and actionable; otherwise this is reasonable to treat as a caller error rather than a problem with the empty head.

Tip: You can customize Greptile's behavior for this repo with .greptile/rules.md and .greptile/config.json.

modules: list[nn.Module] = []
if self.n_layers == 0:
modules.append(nn.Linear(self.embed_dim, self.out_dim))
Expand Down
9 changes: 9 additions & 0 deletions model2vec/train/classifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,15 @@ def classes(self) -> np.ndarray:
"""Return all clasess in the correct order."""
return np.array(self.classes_)

def construct_head(self) -> nn.Sequential:
"""Constructs a classifier head, which always has at least one linear layer."""
if self.n_layers == 0:
linear = nn.Linear(self.embed_dim, self.out_dim)
nn.init.xavier_uniform_(linear.weight)
nn.init.zeros_(linear.bias)
return nn.Sequential(linear)
return super().construct_head()

def predict(
self, X: list[str], show_progress_bar: bool = False, batch_size: int = 1024, threshold: float = 0.5
) -> np.ndarray:
Expand Down
3 changes: 2 additions & 1 deletion model2vec/train/pairs.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,7 +54,8 @@ def __init__(

:param vectors: The embeddings of the staticmodel.
:param tokenizer: The tokenizer.
:param n_layers: The number of layers in the head.
:param n_layers: The number of layers in the head. If this is 0 and `out_dim` equals the embedding
dimension, the model has no head, and the embeddings are used as is.
:param hidden_dim: The hidden dimension of the head.
:param out_dim: The output embedding dimension. If None, defaults to the input embedding dimension.
:param pad_id: The padding id. This is set to 0 in almost all model2vec models.
Expand Down
19 changes: 18 additions & 1 deletion tests/test_export_to_onnx.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@
_save_tokenizer_and_config,
export_model_to_onnx,
)
from model2vec.train import StaticModelForClassification
from model2vec.train import StaticModelForClassification, StaticModelForPairSimilarity


def _tokenize(pipeline: StaticModelPipeline, texts: list[str]) -> tuple[torch.Tensor, torch.Tensor]:
Expand Down Expand Up @@ -147,6 +147,23 @@ def test_pipeline_onnx_matches_projector(
np.testing.assert_allclose(onnx_output, expected, atol=1e-4)


def test_pipeline_onnx_matches_empty_head(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer, tmp_path: Path) -> None:
"""A pipeline whose head has no layers exports the static model's embeddings."""
model = StaticModelForPairSimilarity(
vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0
)
pipeline = model.to_pipeline()
assert pipeline.head.layers == []
texts = ["dog", "cat"]
torch_model = TorchStaticModelPipeline(pipeline)
input_ids, attention_mask = _tokenize(pipeline, texts)

onnx_output = _export(torch_model, input_ids, attention_mask, tmp_path / "model.onnx")
expected = pipeline.predict(texts, use_multiprocessing=False)

np.testing.assert_allclose(onnx_output, expected, atol=1e-4)


def test_save_tokenizer_and_config_removes_post_processor_by_default(
mock_static_model: StaticModel, tmp_path: Path
) -> None:
Expand Down
60 changes: 59 additions & 1 deletion tests/test_trainable.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import logging
from tempfile import TemporaryDirectory
from typing import Any

import numpy as np
import pytest
Expand Down Expand Up @@ -44,7 +45,7 @@ def test_init_base_class(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) ->
"""Test successful initialization of the base class."""
vectors_torched = torch.from_numpy(mock_vectors)
s = BaseFinetuneable(
vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=2, n_layers=0, pad_id=0
vectors=vectors_torched, tokenizer=mock_tokenizer, hidden_dim=256, out_dim=3, n_layers=0, pad_id=0
)
assert s.vectors.shape == mock_vectors.shape
assert s.w.shape[0] == mock_vectors.shape[0]
Expand Down Expand Up @@ -445,6 +446,63 @@ def test_pair_similarity_out_dim_defaults_to_embed_dim(mock_vectors: np.ndarray,
assert s.out_dim == 7


@pytest.mark.parametrize(
"model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity]
)
def test_no_head_without_layers(model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None:
"""Without layers and with an unchanged dimension, the model has no head and matches its static model."""
model = model_class(
vectors=torch.from_numpy(mock_vectors).float(),
tokenizer=mock_tokenizer,
n_layers=0,
out_dim=mock_vectors.shape[1],
)
assert len(model.head) == 0

texts = ["dog cat", "dog"]
np.testing.assert_allclose(model.encode(texts), model.to_static_model().encode(texts), atol=1e-6)
np.testing.assert_allclose(model.encode(texts), model.to_pipeline().predict(texts), atol=1e-6)


@pytest.mark.parametrize(
"model_class", [StaticModelForSimilarity, StaticModelForRegression, StaticModelForPairSimilarity]
)
def test_head_without_layers_changes_dimension(
model_class: Any, mock_vectors: np.ndarray, mock_tokenizer: Tokenizer
) -> None:
"""Without layers but with a different output dimension, the head is a single linear layer."""
model = model_class(
vectors=torch.from_numpy(mock_vectors).float(),
tokenizer=mock_tokenizer,
n_layers=0,
out_dim=mock_vectors.shape[1] + 1,
)
assert len(model.head) == 1
assert isinstance(model.head[0], torch.nn.Linear)


def test_similarity_fit_without_layers(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None:
"""The head of a similarity model follows the dimension of the targets it is fit on."""
model = StaticModelForSimilarity(
vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0
)
texts = ["word1", "word2", "word3", "word1 word2"] * 2
model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1]), max_epochs=1)
assert len(model.head) == 0

model.fit(texts, torch.randn(len(texts), mock_vectors.shape[1] + 1), max_epochs=1)
assert isinstance(model.head[0], torch.nn.Linear)


def test_classifier_keeps_head_when_dimensions_match(mock_vectors: np.ndarray, mock_tokenizer: Tokenizer) -> None:
"""A classifier without layers keeps its linear layer, even if the number of classes equals the dimension."""
model = StaticModelForClassification(
vectors=torch.from_numpy(mock_vectors).float(), tokenizer=mock_tokenizer, n_layers=0
)
assert model.out_dim == mock_vectors.shape[1]
assert isinstance(model.head[0], torch.nn.Linear)


def test_pair_similarity_forward(mock_trained_pair_similarity_pipeline: StaticModelForPairSimilarity) -> None:
"""The forward pass should return one head output per half of the pair batch."""
model = mock_trained_pair_similarity_pipeline
Expand Down
Loading