package router import ( "context" "testing" ) // TestHashEmbedderCyrillic guards the ru-first floor: a byte-only tokenizer // drops every Cyrillic word (bytes ≥ 0x80) and embeds Russian to the zero // vector — cosine 0 across all intents, misrouting every RU utterance. Assert // non-zero vectors, and that shared Russian words produce more similar vectors // than disjoint ones (the point of the bag-of-words floor). func TestHashEmbedderCyrillic(t *testing.T) { e := NewHashEmbedder(1024) ctx := context.Background() nonZero := func(text string) []float32 { v, err := e.Embed(ctx, text) if err != nil { t.Fatalf("embed %q: %v", text, err) } var sum float64 for _, x := range v { sum += float64(x) * float64(x) } if sum == 0 { t.Fatalf("embed %q → zero vector (tokenizer dropped all tokens)", text) } return v } a := nonZero("найди заметку про сервер") b := nonZero("найди заметку про роутер") // shares 3 of 4 words c := nonZero("перезагрузи компьютер") // disjoint if cosine(a, b) <= cosine(a, c) { t.Fatalf("cosine(shared)=%.3f not > cosine(disjoint)=%.3f", cosine(a, b), cosine(a, c)) } } // recordingEmbedder — an asymmetric embedder that only remembers which side // was asked for. Enough to pin the dispatch; real vectors need the model. type recordingEmbedder struct{ calls []string } func (r *recordingEmbedder) Dim() int { return 2 } func (r *recordingEmbedder) Close() error { return nil } func (r *recordingEmbedder) Embed(_ context.Context, _ string) ([]float32, error) { r.calls = append(r.calls, "embed") return []float32{1, 0}, nil } func (r *recordingEmbedder) EmbedQuery(_ context.Context, _ string) ([]float32, error) { r.calls = append(r.calls, "query") return []float32{1, 0}, nil } func (r *recordingEmbedder) EmbedPassage(_ context.Context, _ string) ([]float32, error) { r.calls = append(r.calls, "passage") return []float32{0, 1}, nil } // TestEmbedQueryAndPassageSplit — a question and a stored note must not take // the same path. If both ended up on the same call the asymmetric model buys // nothing, which is the whole reason for the swap. func TestEmbedQueryAndPassageSplit(t *testing.T) { rec := &recordingEmbedder{} if _, err := EmbedQuery(context.Background(), rec, "где логи?"); err != nil { t.Fatalf("EmbedQuery: %v", err) } if _, err := EmbedPassage(context.Background(), rec, "логи в /var/log"); err != nil { t.Fatalf("EmbedPassage: %v", err) } if len(rec.calls) != 2 || rec.calls[0] != "query" || rec.calls[1] != "passage" { t.Errorf("calls %v, want [query passage]", rec.calls) } } // TestEmbedFallsBackToPlainEmbed — HashEmbedder has no sides, so both helpers // must still work and give the same vector. func TestEmbedFallsBackToPlainEmbed(t *testing.T) { h := NewHashEmbedder(64) q, err := EmbedQuery(context.Background(), h, "text") if err != nil { t.Fatalf("EmbedQuery: %v", err) } p, err := EmbedPassage(context.Background(), h, "text") if err != nil { t.Fatalf("EmbedPassage: %v", err) } for i := range q { if q[i] != p[i] { t.Fatalf("hash embedder gave two different vectors for the same text") } } }