Record which embedder wrote the stored vectors and warn on a swap (#378)
The embedder moved from paraphrase-multilingual-MiniLM-L12-v2 to multilingual-e5-small. Both are 384-dimensional, so nothing in the code noticed: cosine between an old stored vector and a new query vector is noise, and recall degrades silently. So the DB now records the embedder that wrote its vectors. One value for the whole DB (migration #11, a small `meta` key/value table) rather than a column on every vector row: the backfill re-embeds every note and fact in one pass, so a per-row marker would hold the same string in every row and cost a column on two tables for nothing. The identity comes from the embedder itself via a new optional ID() method ("multilingual-e5-small@384", model file name plus dimension), so pointing the config at another model changes the string without anyone editing a constant. mavend logs a loud WARNING at startup naming both the stored and the configured embedder when they differ. Detection only — recall behaviour is unchanged. TODO(#378) in store.CheckEmbedder marks where the backfill will hook in. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
This commit is contained in:
@@ -2,6 +2,7 @@ package router
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"math"
|
||||
"unicode"
|
||||
)
|
||||
@@ -19,6 +20,24 @@ type Embedder interface {
|
||||
Close() error
|
||||
}
|
||||
|
||||
// IdentifiedEmbedder — an embedder that can name itself. The name goes into
|
||||
// the DB next to the vectors it wrote, so a later model swap is caught instead
|
||||
// of silently returning nonsense scores (Vikunja #378).
|
||||
type IdentifiedEmbedder interface {
|
||||
Embedder
|
||||
ID() string
|
||||
}
|
||||
|
||||
// EmbedderID is the stable string stored alongside the vectors. It comes from
|
||||
// the embedder itself — nobody hand-types a model name twice — and changes
|
||||
// whenever the model or its dimension changes.
|
||||
func EmbedderID(e Embedder) string {
|
||||
if i, ok := e.(IdentifiedEmbedder); ok {
|
||||
return i.ID()
|
||||
}
|
||||
return fmt.Sprintf("unknown@%d", e.Dim())
|
||||
}
|
||||
|
||||
// AsymmetricEmbedder — an embedder that wants to know whether a text is a
|
||||
// search query or a stored passage. Recall is asymmetric: a short question
|
||||
// goes in, a longer note comes out. The e5 family is trained for exactly that
|
||||
@@ -70,6 +89,10 @@ func NewHashEmbedder(dim int) *HashEmbedder {
|
||||
|
||||
func (h *HashEmbedder) Dim() int { return h.dim }
|
||||
|
||||
// ID names this embedder for the DB marker. The dimension is part of it
|
||||
// because a HashEmbedder of another width is a different vector space.
|
||||
func (h *HashEmbedder) ID() string { return fmt.Sprintf("hash@%d", h.dim) }
|
||||
|
||||
func (h *HashEmbedder) Close() error { return nil }
|
||||
|
||||
func (h *HashEmbedder) Embed(_ context.Context, text string) ([]float32, error) {
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
package router
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestEmbedderIDFromModelPath(t *testing.T) {
|
||||
got := modelIDFromPath("/opt/maven/models/embedder/multilingual-e5-small.onnx")
|
||||
if got != "multilingual-e5-small@384" {
|
||||
t.Fatalf("modelIDFromPath = %q", got)
|
||||
}
|
||||
// A different model file must produce a different id, even at 384 dim.
|
||||
old := modelIDFromPath("/opt/maven/models/embedder/paraphrase-multilingual-MiniLM-L12-v2.onnx")
|
||||
if old == got {
|
||||
t.Fatal("two different models share one id")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmbedderIDIncludesDim(t *testing.T) {
|
||||
if id := EmbedderID(NewHashEmbedder(1024)); id != "hash@1024" {
|
||||
t.Fatalf("EmbedderID = %q", id)
|
||||
}
|
||||
if EmbedderID(NewHashEmbedder(1024)) == EmbedderID(NewHashEmbedder(384)) {
|
||||
t.Fatal("dimension not part of the id")
|
||||
}
|
||||
}
|
||||
@@ -33,6 +33,7 @@ const (
|
||||
type onnxEmbedder struct {
|
||||
tokenizer *unigramTokenizer
|
||||
session *ort.DynamicSession[int64, float32]
|
||||
id string
|
||||
}
|
||||
|
||||
func NewONNXEmbedder(modelPath, tokenizerPath, libPath string) (*onnxEmbedder, error) {
|
||||
@@ -58,11 +59,31 @@ func NewONNXEmbedder(modelPath, tokenizerPath, libPath string) (*onnxEmbedder, e
|
||||
return &onnxEmbedder{
|
||||
tokenizer: tok,
|
||||
session: session,
|
||||
id: modelIDFromPath(modelPath),
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (e *onnxEmbedder) Dim() int { return embedDim }
|
||||
|
||||
// ID names the loaded model for the DB marker (Vikunja #378): the model file's
|
||||
// own name plus the dimension, so pointing the config at another model changes
|
||||
// the string on its own.
|
||||
func (e *onnxEmbedder) ID() string { return e.id }
|
||||
|
||||
// modelIDFromPath turns /opt/.../multilingual-e5-small.onnx into
|
||||
// "multilingual-e5-small@384".
|
||||
func modelIDFromPath(modelPath string) string {
|
||||
name := modelPath
|
||||
if i := strings.LastIndexAny(name, "/\\"); i >= 0 {
|
||||
name = name[i+1:]
|
||||
}
|
||||
name = strings.TrimSuffix(name, ".onnx")
|
||||
if name == "" {
|
||||
name = "onnx"
|
||||
}
|
||||
return fmt.Sprintf("%s@%d", name, embedDim)
|
||||
}
|
||||
|
||||
// Embed treats the text as a query. The classifier compares one short
|
||||
// utterance to another short seed phrase, so both sides get the same prefix
|
||||
// and the comparison stays fair. The recall path must call EmbedQuery and
|
||||
|
||||
Reference in New Issue
Block a user