Ship voice enrolment, and report recognition as blocked (#255)
Maven can now be told who someone is. She cannot yet tell who is speaking, and this commit is careful to say so rather than pretend otherwise. What works: profiles are enrolled from several deliberately recorded samples, listed, and deleted. They live in the existing memory_vectors table under a "speaker:" id prefix, so there is no migration; what that needed was a wider interface than memory.Store, hence memory.Catalog with ByPrefix and Delete. Delete is the load-bearing half — a voiceprint someone asked to be rid of has to actually go, and a search-only store cannot do that. InMemoryStore.Insert became an upsert by id to match what the persistent store already did. What does not work, and why it is not faked: there is no speaker-embedding model on this box. Sixteen ggufs in /mnt/hdd1/llms, all text; no ECAPA, no x-vector, no titanet, no wespeaker, no .onnx anywhere under /mnt/hdd1. So newSpeakerEmbedder returns nil, internal/speaker falls back to speaker.Disabled, Identify answers ErrDisabled, and the daemon logs which half is off at startup. The plan's "simple MFCC + GMM" floor is refused in the package comment: MFCC cosine distance detects channel and loudness as much as voice, and a biometric that is confidently wrong writes false claims about named people into his memory. A bad floor is worse than none here. Refused as well, and the reason is in enroll.go's doc comment: the plan asked for unknown speakers to be enrolled on first interaction with a TTS "кто это?". There is no request shape in the protocol that could express that. Taking a biometric of whoever walks past the microphone does it to guests who are not party to the exchange, and a synthesised question into a room is not consent from whoever answers. Authority: enrolment is AuthStepUp, because it is a deliberate sit-down act that writes a biometric of a named person and never something done by voice mid-conversation. Deletion is one rung lower at AuthWrite, deliberately inverting the usual pattern — getting rid of a biometric must never be the harder half. Listing is AuthRead and never returns the vectors themselves. Off unless configured: no speaker block means the three methods answer ErrUnknownMethod, so a default box has no wire path that takes a voiceprint. make build and make test pass. Vikunja #255 Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TrVSBKe3RFDF4fGYKWYQnX
This commit is contained in:
@@ -340,6 +340,10 @@ func run(args []string) error {
|
||||
// retention loop. Off unless a capture block enables it, in which case
|
||||
// all four capture methods answer ErrUnknownMethod.
|
||||
wireCapture(srv, keeper, st, voiceW, phr, cfg)
|
||||
// Voice identification (Vikunja #255). Enrolment plumbing only until a
|
||||
// speaker-embedding model exists on disk; off entirely without a speaker
|
||||
// block, so no wire path takes a voiceprint on a default box.
|
||||
wireSpeaker(srv, st, cfg)
|
||||
}
|
||||
|
||||
// WrapKeyFn — wraps the env key with a passkey credential public key and
|
||||
@@ -485,6 +489,10 @@ func run(args []string) error {
|
||||
wireModelSwap(srv, phr, cfg)
|
||||
keeper := wireVision(ctx, srv, st, embedderOf(voiceW), cfg)
|
||||
wireCapture(srv, keeper, st, voiceW, phr, cfg)
|
||||
// Voice identification (Vikunja #255). Enrolment plumbing only until a
|
||||
// speaker-embedding model exists on disk; off entirely without a speaker
|
||||
// block, so no wire path takes a voiceprint on a default box.
|
||||
wireSpeaker(srv, st, cfg)
|
||||
|
||||
// Start voice server.
|
||||
if voiceW != nil {
|
||||
|
||||
@@ -0,0 +1,146 @@
|
||||
// mavend/speaker.go — core's half of voice identification (Vikunja #255,
|
||||
// docs/plans/10-speaker-recognition.md).
|
||||
//
|
||||
// # What is actually wired here, and what is not
|
||||
//
|
||||
// The enrolment plumbing is real: profiles are stored, listed and deleted, and
|
||||
// the wire methods exist as soon as a speaker block is configured. The
|
||||
// recognising half is NOT, and cannot be on this box, because there is no
|
||||
// speaker-embedding model on disk — no ECAPA, no x-vector, no titanet, no
|
||||
// wespeaker, nothing in /mnt/hdd1/llms but text ggufs. Until one is downloaded,
|
||||
// newSpeakerEmbedder returns nil, internal/speaker falls back to
|
||||
// speaker.Disabled, and every Identify answers ErrDisabled. The daemon logs
|
||||
// which half is off at startup rather than pretending.
|
||||
//
|
||||
// This is deliberately not papered over with a hand-rolled MFCC floor. A
|
||||
// biometric that is confidently wrong writes false claims about named people
|
||||
// into his memory, and that is worse than a capability that is honestly absent.
|
||||
//
|
||||
// # Off unless configured
|
||||
//
|
||||
// No speaker block, or one without enabled, ⇒ the three methods do not exist and
|
||||
// answer ErrUnknownMethod. On an unconfigured box there is no wire path that
|
||||
// takes a voiceprint at all.
|
||||
//
|
||||
// # The refused design step
|
||||
//
|
||||
// The plan asks for unknown speakers to be enrolled on first interaction. That
|
||||
// is refused in internal/speaker/enroll.go and there is no handler for it here:
|
||||
// no request shape in the protocol enrols whoever just spoke. Taking a biometric
|
||||
// of a guest who walked past the microphone is not something this daemon does.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"log"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/ipc"
|
||||
"github.com/kami/maven/internal/speaker"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
// speakerWiring holds the recognizer behind the three IPC handlers.
|
||||
type speakerWiring struct {
|
||||
rec *speaker.Recognizer
|
||||
}
|
||||
|
||||
// newSpeakerEmbedder loads the speaker-embedding model named by the config.
|
||||
//
|
||||
// It always returns nil today. The seam exists so that wiring a real model is a
|
||||
// change to this one function and nothing else: give it a loader, and Identify
|
||||
// starts working with no change to the store, the protocol, the auth table or
|
||||
// the handlers. See the plan document for what to download.
|
||||
func newSpeakerEmbedder(cfg *config.SpeakerConfig) speaker.Embedder {
|
||||
if cfg == nil || cfg.ModelPath == "" {
|
||||
return nil
|
||||
}
|
||||
log.Printf("speaker: model_path %q is configured but no embedding backend is built yet; "+
|
||||
"enrolment and deletion work, recognition does not (Vikunja #255)", cfg.ModelPath)
|
||||
return nil
|
||||
}
|
||||
|
||||
// newSpeakerWiring builds the recognizer, or nil when the capability is off.
|
||||
func newSpeakerWiring(st *store.Store, cfg *config.Config) *speakerWiring {
|
||||
if cfg == nil || cfg.Speaker == nil || !cfg.Speaker.Enabled {
|
||||
return nil
|
||||
}
|
||||
if st == nil {
|
||||
log.Print("speaker: enabled but there is no store to keep profiles in; staying off")
|
||||
return nil
|
||||
}
|
||||
rec, err := speaker.New(newSpeakerEmbedder(cfg.Speaker), st.VectorMemory(), speaker.Config{
|
||||
Threshold: cfg.Speaker.Threshold,
|
||||
MinSeconds: cfg.Speaker.MinSeconds,
|
||||
})
|
||||
if err != nil {
|
||||
log.Printf("speaker: %v; staying off", err)
|
||||
return nil
|
||||
}
|
||||
if rec.Enabled() {
|
||||
log.Printf("speaker: recognition on, threshold %.2f", rec.Threshold())
|
||||
} else {
|
||||
log.Print("speaker: enrolment on, recognition BLOCKED — no speaker-embedding model " +
|
||||
"on this box (see docs/plans/10-speaker-recognition.md)")
|
||||
}
|
||||
return &speakerWiring{rec: rec}
|
||||
}
|
||||
|
||||
func (w *speakerWiring) enroll(ctx context.Context, req ipc.EnrollSpeakerReq) (ipc.EnrollSpeakerResp, error) {
|
||||
p, err := w.rec.Enroll(ctx, req.ID, req.Name, req.Samples)
|
||||
if err != nil {
|
||||
return ipc.EnrollSpeakerResp{}, speakerErr(err)
|
||||
}
|
||||
return ipc.EnrollSpeakerResp{Speaker: toWireSpeaker(p)}, nil
|
||||
}
|
||||
|
||||
func (w *speakerWiring) list(ctx context.Context) (ipc.ListSpeakersResp, error) {
|
||||
ps, err := w.rec.List(ctx)
|
||||
if err != nil {
|
||||
return ipc.ListSpeakersResp{}, speakerErr(err)
|
||||
}
|
||||
out := make([]ipc.Speaker, 0, len(ps))
|
||||
for _, p := range ps {
|
||||
out = append(out, toWireSpeaker(p))
|
||||
}
|
||||
return ipc.ListSpeakersResp{Speakers: out, Enabled: w.rec.Enabled()}, nil
|
||||
}
|
||||
|
||||
func (w *speakerWiring) forget(ctx context.Context, req ipc.ForgetSpeakerReq) error {
|
||||
return speakerErr(w.rec.Forget(ctx, req.ID))
|
||||
}
|
||||
|
||||
// toWireSpeaker drops the voiceprint. A listing says who is enrolled; it does
|
||||
// not hand the biometric back out over the socket.
|
||||
func toWireSpeaker(p speaker.Profile) ipc.Speaker {
|
||||
return ipc.Speaker{ID: p.ID, Name: p.Name, Enrolled: p.Enrolled, Samples: p.Samples}
|
||||
}
|
||||
|
||||
// speakerErr maps the package sentinels onto the wire vocabulary so a surface
|
||||
// can tell "you asked wrong" from "core broke".
|
||||
func speakerErr(err error) error {
|
||||
switch {
|
||||
case err == nil:
|
||||
return nil
|
||||
case errors.Is(err, speaker.ErrNotFound):
|
||||
return ipc.ErrNoFact
|
||||
case errors.Is(err, speaker.ErrBadID),
|
||||
errors.Is(err, speaker.ErrBadFormat),
|
||||
errors.Is(err, speaker.ErrTooShort):
|
||||
return errors.Join(ipc.ErrBadParams, err)
|
||||
default:
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
// wireSpeaker attaches the three handlers when the capability is configured.
|
||||
func wireSpeaker(srv *ipc.Server, st *store.Store, cfg *config.Config) {
|
||||
w := newSpeakerWiring(st, cfg)
|
||||
if w == nil {
|
||||
return
|
||||
}
|
||||
srv.EnrollSpeakerFn = w.enroll
|
||||
srv.ListSpeakersFn = w.list
|
||||
srv.ForgetSpeakerFn = w.forget
|
||||
}
|
||||
@@ -1,27 +1,125 @@
|
||||
# Plan: Speaker Recognition
|
||||
|
||||
**Goal:** Maven can distinguish between different speakers on the voice channel — recognize known voices (the user, family members) and tag facts/notes/transcripts with a speaker identity.
|
||||
**Goal:** Maven can tell who is speaking on the voice channel, and tag what she writes with
|
||||
who said it.
|
||||
|
||||
**Done when:**
|
||||
- Speaker embedding extractor (e.g., ECAPA-TDNN or a simple MFCC + GMM) runs on incoming voice PCM before STT
|
||||
- Embedding is compared against enrolled speaker profiles (stored as vectors in the `memory_vectors` table alongside semantic memory)
|
||||
- Unknown speakers are enrolled on first interaction (prompt: "кто это?")
|
||||
- All voice fact/note writes are tagged with `speaker:<id>` in the value/source metadata
|
||||
- Speaker identity is available as context to the router, phraser, and replier ("ok, <name>")
|
||||
**Status (2026-08-01, Vikunja #255):** the enrolment half is shipped. The recognising half is
|
||||
**BLOCKED on a model download** — there is no speaker-embedding model on this box, and one
|
||||
was not invented to fill the gap. See "Blocked, and on what" below.
|
||||
|
||||
**Scope:**
|
||||
- New `internal/speaker/` package — enrollment, recognition, embedding extraction
|
||||
- Reuses `internal/store.MemoryStore` for speaker vector storage (same `memory_vectors` table, different `source` prefix)
|
||||
- Reuses `internal/audio` for PCM preprocessing
|
||||
- Integration point: `cmd/mavend/voice.go:HandlePushToTalk` — speaker ID extracted before STT, passed through context
|
||||
## What shipped
|
||||
|
||||
**Steps:**
|
||||
1. Research speaker embedding approaches — simplest floor: MFCC + cosine similarity via `github.com/mjibson/go-dsp` or a pre-trained ONNX model (SpeechBrain ECAPA)
|
||||
2. Create `internal/speaker/recognizer.go` — `Recognizer` interface: `Identify(pcm []float32) (SpeakerID, confidence)`, `Enroll(id, pcm)`
|
||||
3. Create `internal/speaker/store.go` — speaker profile CRUD via `store.MemoryStore`: `Insert("speaker:<id>", embedding, meta)`, `Search(embedding, k)`
|
||||
4. Create `internal/speaker/enroll.go` — enrollment flow: capture N seconds of audio, extract embedding, prompt for name via TTS + STT round-trip
|
||||
5. Wire into `cmd/mavend/voice.go:HandlePushToTalk` — run speaker ID on the PCM before STT; pass speaker ID through `context.Context` to `applyAction`
|
||||
6. Tag all voice-written facts/notes with speaker ID — `Source` becomes `tap:voice:speaker:<id>` or metadata field
|
||||
7. Add IPC methods `MethodEnrollSpeaker`, `MethodListSpeakers`, `MethodRemoveSpeaker`
|
||||
8. Add speaker config block to `voice` in `config.Config` — `{speaker_recognition: true, model_path}`
|
||||
9. Test with 2+ recorded voice samples — verify correct identification and rejection of unknown speakers
|
||||
| Piece | Where | State |
|
||||
|---|---|---|
|
||||
| `Recognizer` — identify, list, get, forget | `internal/speaker/recognizer.go` | done; `Identify` answers `ErrDisabled` until a model exists |
|
||||
| Enrolment — several samples, averaged, re-normalised | `internal/speaker/enroll.go` | done |
|
||||
| Profile shape, id validation, cosine similarity | `internal/speaker/speaker.go` | done |
|
||||
| Profile storage as `speaker:<id>` vectors | `internal/memory` `Catalog` + `internal/store/memory.go` | done, no schema migration |
|
||||
| Config block, off by default | `internal/config` `SpeakerConfig` | done |
|
||||
| `enroll_speaker` / `list_speakers` / `forget_speaker` | `internal/ipc` | done, absent unless configured |
|
||||
| Authority rows | `internal/auth/policy.go` | done — enrol step-up, forget write, list read |
|
||||
| Daemon wiring + honest startup log | `cmd/mavend/speaker.go` | done |
|
||||
| Embedding backend | `newSpeakerEmbedder` | **BLOCKED** — returns nil, seam only |
|
||||
| Tagging voice writes with the speaker | `cmd/mavend/voice.go` | not wired; nothing to tag with yet |
|
||||
|
||||
## Blocked, and on what
|
||||
|
||||
A voiceprint needs a speaker-embedding model. The box was searched: `/mnt/hdd1/llms` holds
|
||||
sixteen ggufs across seven families and every one of them is a text model. There is no ECAPA,
|
||||
no x-vector, no titanet, no wespeaker, and no `.onnx` under `/mnt/hdd1` at all. There are also
|
||||
no enrolment samples, because nothing has ever recorded any.
|
||||
|
||||
To unblock, two things are needed and neither can be done from inside the repo:
|
||||
|
||||
1. **A model.** SpeechBrain ECAPA-TDNN exported to ONNX (`speechbrain/spkrec-ecapa-voxceleb`,
|
||||
192-dim) is the usual choice and runs on CPU in well under a second for a few seconds of
|
||||
audio. Download it per the recipe in `AGENTS.md`, put it beside the other models so the
|
||||
bind mount picks it up, and point `speaker.model_path` at it.
|
||||
2. **An implementation of one function.** `newSpeakerEmbedder` in `cmd/mavend/speaker.go` is
|
||||
the entire seam: give it an ONNX session that turns `audio.Audio` into a `[]float32` and
|
||||
`Identify` starts working. Nothing else changes — not the store, not the protocol, not the
|
||||
authority table, not the handlers. `internal/onnx` already loads the e5 embedder, so the
|
||||
runtime wiring exists to copy.
|
||||
3. **Enrolment samples**, three or more per person, recorded deliberately.
|
||||
|
||||
### Why there is no fallback
|
||||
|
||||
The original plan offered "a simple MFCC + GMM" as the floor. That is refused. MFCC cosine
|
||||
distance is a channel and loudness detector as much as a voice detector: it will happily match
|
||||
two different people who sit at the same distance from the same microphone, and it drifts when
|
||||
the room changes. A general classifier that is sometimes wrong is a nuisance; a **biometric**
|
||||
that is confidently wrong writes false claims about named people into his memory, and then
|
||||
those claims get recalled as fact. For this capability a bad floor is worse than none, so the
|
||||
shipped state is honest absence: `speaker.Disabled`, `ErrDisabled`, and a startup line saying
|
||||
so.
|
||||
|
||||
## The refusals, and why
|
||||
|
||||
- **Unknown speakers are NOT enrolled on first interaction.** The plan's fourth "done when"
|
||||
bullet asked for exactly that, with a TTS "кто это?" prompt. It is refused in
|
||||
`enroll.go`'s doc comment and there is no request shape in the protocol that could express
|
||||
it. Enrolling a voice is taking a biometric of a person; doing it automatically to whoever
|
||||
walks past the microphone does it to guests who are not party to the exchange, and a
|
||||
synthesised question into a room is not consent from whoever happens to answer. Enrolment is
|
||||
an explicit act: an id, a name, and samples recorded for the purpose.
|
||||
- **One sample is not enough.** Three separate utterances and nine seconds minimum. A profile
|
||||
built from one sentence encodes that sentence as much as the person, and the threshold then
|
||||
behaves unpredictably against everything else.
|
||||
- **An unknown voice stays unknown.** Below threshold, `Identify` returns `ErrUnknown` naming
|
||||
the closest profile in the error text for diagnosis, never as an answer. Guessing who is in
|
||||
the room is how false memories about people get written.
|
||||
- **Deletion is one authority rung below enrolment.** Everywhere else in `policy.go` the
|
||||
destructive direction is gated at least as hard as the constructive one. Here that would be
|
||||
backwards: getting rid of a biometric must never be the harder half.
|
||||
- **The voiceprint never crosses the socket.** `ListSpeakersResp` carries ids, names, dates
|
||||
and sample counts. The vector stays in core.
|
||||
- **Off unless configured.** No `speaker` block ⇒ the three methods answer
|
||||
`ErrUnknownMethod`. There is no wire path on a default box that takes a voiceprint.
|
||||
|
||||
## Storage
|
||||
|
||||
Profiles live in the existing `memory_vectors` table under the `speaker:` id prefix, as the
|
||||
plan intended, so there is no migration. What that needed was a wider interface than
|
||||
`memory.Store`: `memory.Catalog` adds `ByPrefix` and `Delete`. `Delete` is the load-bearing
|
||||
one — a voiceprint someone asked to be rid of has to actually go, and a search-only store
|
||||
cannot do that. `InMemoryStore.Insert` also became an upsert by id, matching what the
|
||||
persistent store already did, so re-enrolling replaces a profile instead of stacking a second
|
||||
one behind the first.
|
||||
|
||||
Profiles do not collide with note or fact vectors: they are only ever read through
|
||||
`ByPrefix("speaker:")`, and a note search never returns one because the prefix is not in its
|
||||
query path.
|
||||
|
||||
## Config
|
||||
|
||||
```json
|
||||
"speaker": {
|
||||
"enabled": true,
|
||||
"model_path": "/opt/maven/models/spk/ecapa-voxceleb.onnx",
|
||||
"lib_path": "/opt/maven/lib",
|
||||
"threshold": 0.7,
|
||||
"min_seconds": 2.0
|
||||
}
|
||||
```
|
||||
|
||||
`Recognizes()` requires both `enabled` and a `model_path`, so a half-filled block reads as off
|
||||
rather than as a capability that fails every turn. With `enabled` and no model the daemon still
|
||||
attaches the three methods — profiles can be created, listed and deleted — and logs that
|
||||
recognition is blocked.
|
||||
|
||||
## Still open
|
||||
|
||||
- The embedding backend (above). Everything below waits on it.
|
||||
- **Tagging voice writes.** `Profile.Source("tap:voice")` already produces
|
||||
`tap:voice:speaker:kami`, which is the shape step 6 asked for, but nothing calls it yet:
|
||||
with no recogniser there is no id to tag with. When the model lands, the hook is in the
|
||||
voice path before STT.
|
||||
- **Speaker as router/phraser context.** Same dependency. Note the persona constraint when it
|
||||
arrives: Maven addresses the owner informally and speaks to him, so "ok, <name>" needs care
|
||||
for anyone who is not him.
|
||||
- **An enrolment surface.** The three IPC methods exist; no page drives them. Enrolment is
|
||||
step-up, so it belongs on `/dash` behind a passkey, with a per-profile forget button next to
|
||||
each row — that button is the reason `list_speakers` exists.
|
||||
- **A speaker column on the meeting recorder** (#253). Attributing lines in a transcript is
|
||||
the obvious pairing, and it is the place where getting attribution wrong is most damaging,
|
||||
so it waits for a real model too.
|
||||
|
||||
@@ -442,3 +442,33 @@ func TestRequirement_Capture(t *testing.T) {
|
||||
t.Errorf("voice starting a capture = %v; want allowed", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRequirement_Speaker — a voiceprint is a biometric of a named person, so
|
||||
// taking one is step-up: a deliberate act from a surface that can carry a
|
||||
// passkey gesture, never something the voice path does mid-conversation.
|
||||
//
|
||||
// Deletion is one rung lower, and that asymmetry is the point. Everywhere else
|
||||
// in the table the destructive direction is gated at least as hard as the
|
||||
// constructive one; for a biometric that would be backwards, because getting
|
||||
// rid of it must never be the harder half.
|
||||
func TestRequirement_Speaker(t *testing.T) {
|
||||
if got := Requirement(ipc.MethodEnrollSpeaker); got != AuthStepUp {
|
||||
t.Errorf("EnrollSpeaker authority = %v; want AuthStepUp", got)
|
||||
}
|
||||
if got := Requirement(ipc.MethodForgetSpeaker); got != AuthWrite {
|
||||
t.Errorf("ForgetSpeaker authority = %v; want AuthWrite", got)
|
||||
}
|
||||
if got := Requirement(ipc.MethodListSpeakers); got != AuthRead {
|
||||
t.Errorf("ListSpeakers authority = %v; want AuthRead", got)
|
||||
}
|
||||
// Voice cannot enrol anybody, however the utterance is phrased.
|
||||
voice := Scope{Surface: SurfaceVoice, Module: "voice", SourceScope: []string{"*"}}
|
||||
if err := Can(ipc.MethodEnrollSpeaker, voice, nil); err == nil {
|
||||
t.Error("voice enrolling a speaker was allowed; want refused")
|
||||
}
|
||||
// But it can read the roster, which is what answering "кого ты знаешь?"
|
||||
// needs.
|
||||
if err := Can(ipc.MethodListSpeakers, voice, nil); err != nil {
|
||||
t.Errorf("voice listing speakers = %v; want allowed", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -75,6 +75,22 @@ func Requirement(m ipc.Method) Authority {
|
||||
// exist at all unless the operator enabled a capture block, and no
|
||||
// recording can begin without someone saying so.
|
||||
return AuthWrite
|
||||
case ipc.MethodEnrollSpeaker:
|
||||
// Taking a voiceprint (Vikunja #255). AuthStepUp, and unlike recording a
|
||||
// meeting there is no reason to soften it: enrolment is not a thing anyone
|
||||
// does by voice mid-conversation. It is a deliberate sit-down with a
|
||||
// surface that can carry a passkey gesture, and it writes a biometric of a
|
||||
// named person. If the gesture is inconvenient, that is the correct amount
|
||||
// of friction for this particular write.
|
||||
return AuthStepUp
|
||||
case ipc.MethodForgetSpeaker:
|
||||
// Deleting a voiceprint. One rung BELOW enrolment on purpose. Everywhere
|
||||
// else in this table the destructive direction is gated at least as hard
|
||||
// as the constructive one, and here that would be wrong: getting rid of a
|
||||
// biometric must never be the harder half. The worst a caller at this rung
|
||||
// can do is make Maven stop recognising someone, which is the state the
|
||||
// box ships in anyway.
|
||||
return AuthWrite
|
||||
case ipc.MethodWriteFact:
|
||||
return AuthWrite
|
||||
case ipc.MethodAssertStepUp:
|
||||
@@ -113,6 +129,10 @@ func Requirement(m ipc.Method) Authority {
|
||||
// "что ты записываешь?" — the read side of the recorder. It reports a
|
||||
// label, a start time and a byte count, begins nothing and keeps nothing.
|
||||
ipc.MethodCaptureStatus,
|
||||
// Who is enrolled. Returns ids, names and enrolment dates — never the
|
||||
// voiceprints themselves, which stay in core. Listing the people Maven can
|
||||
// recognise is exactly the read a surface needs to offer a "forget" button.
|
||||
ipc.MethodListSpeakers,
|
||||
// The read side of the model swap: which model is resident, which ones are
|
||||
// allowlisted. It loads nothing and changes nothing.
|
||||
ipc.MethodModelStatus:
|
||||
|
||||
@@ -212,6 +212,11 @@ type Config struct {
|
||||
// default. See CaptureConfig.
|
||||
Capture *CaptureConfig `json:"capture,omitempty"`
|
||||
|
||||
// Speaker — voice identification (Vikunja #255). nil / absent ⇒ no
|
||||
// voiceprint is ever computed and nobody can be enrolled. Enabling it needs
|
||||
// a speaker-embedding model, which is not on this box. See SpeakerConfig.
|
||||
Speaker *SpeakerConfig `json:"speaker,omitempty"`
|
||||
|
||||
// MCP — Model Context Protocol servers Maven connects OUT to (Vikunja
|
||||
// #251). nil / absent / no enabled server ⇒ no connection is made and no
|
||||
// tool is discovered, like every other capability that reaches outside the
|
||||
@@ -619,6 +624,45 @@ func (c *CaptureConfig) MaxDuration() time.Duration {
|
||||
return time.Duration(c.MaxMinutes) * time.Minute
|
||||
}
|
||||
|
||||
// SpeakerConfig — voice identification (internal/speaker,
|
||||
// docs/plans/10-speaker-recognition.md).
|
||||
//
|
||||
// Absent, or enabled=false, ⇒ no voiceprint is computed for any turn, the
|
||||
// enrolment methods do not exist, and nobody can be enrolled. A voiceprint is
|
||||
// biometric data about a person, so this one is off until someone typed a model
|
||||
// path on purpose.
|
||||
//
|
||||
// It cannot currently be turned on: there is no speaker-embedding model on this
|
||||
// box. See the plan document for what to download.
|
||||
type SpeakerConfig struct {
|
||||
// Enabled — may she work out who is speaking. Default false.
|
||||
Enabled bool `json:"enabled,omitempty"`
|
||||
|
||||
// ModelPath — an ECAPA-TDNN (or equivalent) speaker-embedding ONNX model.
|
||||
// Required; without it the recognizer runs disabled and says so once.
|
||||
ModelPath string `json:"model_path,omitempty"`
|
||||
|
||||
// LibPath — onnxruntime shared library, as for the text embedder. Empty ⇒
|
||||
// the same default the embedder block uses.
|
||||
LibPath string `json:"lib_path,omitempty"`
|
||||
|
||||
// Threshold — cosine similarity a match must beat. 0 ⇒
|
||||
// speaker.DefaultThreshold (0.7). Lower it and she starts calling guests by
|
||||
// his name, which is the expensive direction of this error.
|
||||
Threshold float64 `json:"threshold,omitempty"`
|
||||
|
||||
// MinSeconds — least speech an identification will look at. 0 ⇒
|
||||
// speaker.DefaultMinSeconds (2s).
|
||||
MinSeconds float64 `json:"min_seconds,omitempty"`
|
||||
}
|
||||
|
||||
// Recognizes reports whether voice identification should be wired. Safe on a
|
||||
// nil receiver, and false without a model path — enabled with nothing to embed
|
||||
// with is a misconfiguration, not a capability.
|
||||
func (s *SpeakerConfig) Recognizes() bool {
|
||||
return s != nil && s.Enabled && strings.TrimSpace(s.ModelPath) != ""
|
||||
}
|
||||
|
||||
// WeatherConfig configures the weather provider for voice queries.
|
||||
type WeatherConfig struct {
|
||||
Provider string `json:"provider,omitempty"` // "open-meteo" or "" → stub
|
||||
|
||||
@@ -22,6 +22,9 @@ func TestSensesOffByDefault(t *testing.T) {
|
||||
if cfg.Capture.MaxDuration() != 0 {
|
||||
t.Error("a nil capture block invented a duration")
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("speaker recognition is on with no speaker block")
|
||||
}
|
||||
}
|
||||
|
||||
// The recorder is the capability that most needs its default to be off, so it
|
||||
@@ -158,3 +161,63 @@ func TestMediaWithoutVisionIsValid(t *testing.T) {
|
||||
t.Error("vision came on by itself")
|
||||
}
|
||||
}
|
||||
|
||||
// A voiceprint is a biometric of a named person. Nothing about it turns on by
|
||||
// itself: no speaker block means no recognition, and no enrolment either.
|
||||
func TestSpeakerIsOffUntilExplicitlyEnabled(t *testing.T) {
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(`{}`), &cfg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("speaker recognition came on with no config at all")
|
||||
}
|
||||
var empty Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{}}`), &empty); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if empty.Speaker.Recognizes() {
|
||||
t.Error("an empty speaker block enabled recognition")
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled alone is not enough: recognition needs a model, and on this box there
|
||||
// is none. Recognizes() must stay false so the daemon reports the honest state
|
||||
// instead of claiming a capability it cannot perform.
|
||||
func TestSpeakerNeedsBothEnabledAndAModel(t *testing.T) {
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{"enabled":true}}`), &cfg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("enabled with no model_path claimed to recognise")
|
||||
}
|
||||
var only Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{"model_path":"/opt/x.onnx"}}`), &only); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if only.Speaker.Recognizes() {
|
||||
t.Error("a model_path alone enabled recognition")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpeakerBlockParsesFromJSON(t *testing.T) {
|
||||
const raw = `{"speaker":{"enabled":true,"model_path":"/opt/maven/models/spk/ecapa.onnx",` +
|
||||
`"lib_path":"/opt/maven/lib","threshold":0.62,"min_seconds":1.5}}`
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(raw), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if !cfg.Speaker.Recognizes() {
|
||||
t.Fatal("speaker did not parse as enabled")
|
||||
}
|
||||
if cfg.Speaker.ModelPath != "/opt/maven/models/spk/ecapa.onnx" {
|
||||
t.Errorf("model_path = %q", cfg.Speaker.ModelPath)
|
||||
}
|
||||
if cfg.Speaker.LibPath != "/opt/maven/lib" {
|
||||
t.Errorf("lib_path = %q", cfg.Speaker.LibPath)
|
||||
}
|
||||
if cfg.Speaker.Threshold != 0.62 || cfg.Speaker.MinSeconds != 1.5 {
|
||||
t.Errorf("thresholds = %+v", cfg.Speaker)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -301,6 +301,51 @@ type CaptureStatusResp struct {
|
||||
Bytes int `json:"bytes,omitempty"`
|
||||
}
|
||||
|
||||
// EnrollSpeakerReq — register a voice (Vikunja #255).
|
||||
//
|
||||
// Samples are separate utterances recorded deliberately for this purpose, not
|
||||
// audio harvested from ordinary turns. internal/speaker requires several of
|
||||
// them totalling enough seconds, and refuses one long clip: a profile built
|
||||
// from a single sentence encodes that sentence as much as the person.
|
||||
//
|
||||
// There is no "enrol whoever just spoke" request shape, and that omission is
|
||||
// the point. Taking a biometric of a guest because they walked past the
|
||||
// microphone is not something a wire protocol should make easy.
|
||||
type EnrollSpeakerReq struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name,omitempty"`
|
||||
Samples []audio.Audio `json:"samples"`
|
||||
}
|
||||
|
||||
// Speaker — one enrolled voice as a surface sees it. The voiceprint itself is
|
||||
// never sent: a listing says who is enrolled, it does not hand out the
|
||||
// biometric.
|
||||
type Speaker struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Enrolled time.Time `json:"enrolled"`
|
||||
Samples int `json:"samples"`
|
||||
}
|
||||
|
||||
// EnrollSpeakerResp — the profile that was written.
|
||||
type EnrollSpeakerResp struct {
|
||||
Speaker Speaker `json:"speaker"`
|
||||
}
|
||||
|
||||
// ListSpeakersResp — who is enrolled, sorted by id. Enabled is false when no
|
||||
// embedding model is wired, which is this box's state: the profiles can be
|
||||
// listed and deleted, nothing can be recognised.
|
||||
type ListSpeakersResp struct {
|
||||
Speakers []Speaker `json:"speakers"`
|
||||
Enabled bool `json:"enabled"`
|
||||
}
|
||||
|
||||
// ForgetSpeakerReq — delete one voiceprint. This is the request that must
|
||||
// always work; a biometric someone asked to be rid of has to actually go.
|
||||
type ForgetSpeakerReq struct {
|
||||
ID string `json:"id"`
|
||||
}
|
||||
|
||||
// SwapModelReq — load another resident model without restarting the daemon
|
||||
// (Vikunja #250). ModelPath must be one of the paths in phraser.swap_models;
|
||||
// anything else is ErrForbidden, and an unconfigured allowlist makes the whole
|
||||
|
||||
@@ -515,6 +515,34 @@ func (c *Client) CaptureStatus(ctx context.Context) (CaptureStatusResp, error) {
|
||||
return r, nil
|
||||
}
|
||||
|
||||
// EnrollSpeaker registers a voice from several deliberately recorded samples
|
||||
// (Vikunja #255). ErrUnknownMethod means no speaker block is configured, which
|
||||
// is the default: on an unconfigured box there is no way to take a voiceprint.
|
||||
func (c *Client) EnrollSpeaker(ctx context.Context, req EnrollSpeakerReq) (EnrollSpeakerResp, error) {
|
||||
var r EnrollSpeakerResp
|
||||
if err := c.call(ctx, MethodEnrollSpeaker, req, &r); err != nil {
|
||||
return EnrollSpeakerResp{}, err
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
|
||||
// ListSpeakers reports who is enrolled. The voiceprints themselves stay in
|
||||
// core. Enabled is false when profiles exist but no embedding model is wired,
|
||||
// so a surface can say "enrolled, not recognising" rather than implying Maven
|
||||
// knows who is talking.
|
||||
func (c *Client) ListSpeakers(ctx context.Context) (ListSpeakersResp, error) {
|
||||
var r ListSpeakersResp
|
||||
if err := c.call(ctx, MethodListSpeakers, nil, &r); err != nil {
|
||||
return ListSpeakersResp{}, err
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
|
||||
// ForgetSpeaker deletes one voiceprint.
|
||||
func (c *Client) ForgetSpeaker(ctx context.Context, id string) error {
|
||||
return c.call(ctx, MethodForgetSpeaker, ForgetSpeakerReq{ID: id}, nil)
|
||||
}
|
||||
|
||||
// SwapModel asks core to load another resident model (Vikunja #250).
|
||||
// ErrUnknownMethod means core has no phraser.swap_models allowlist configured;
|
||||
// ErrForbidden means the path is not on it, or step-up was not asserted. A
|
||||
|
||||
@@ -473,6 +473,13 @@ type Server struct {
|
||||
CaptureStopFn CaptureStopFunc
|
||||
CaptureStatusFn CaptureStatusFunc
|
||||
|
||||
// Speaker* — voice identification (Vikunja #255). Set by the daemon only
|
||||
// when a speaker block is configured; nil ⇒ all three methods answer
|
||||
// ErrUnknownMethod, so on an unconfigured box no wire path enrols a voice.
|
||||
EnrollSpeakerFn EnrollSpeakerFunc
|
||||
ListSpeakersFn ListSpeakersFunc
|
||||
ForgetSpeakerFn ForgetSpeakerFunc
|
||||
|
||||
// UnlockFn — unwraps the store encryption key from the wrapped blob using
|
||||
// the passkey credential public key, opens the encrypted store, and wires
|
||||
// the rest of the daemon (voice, loop, delivery). Set by the daemon when
|
||||
@@ -510,6 +517,12 @@ type CaptureAppendFunc func(ctx context.Context, req CaptureAppendReq) (CaptureA
|
||||
type CaptureStopFunc func(ctx context.Context, req CaptureStopReq) (CaptureStopResp, error)
|
||||
type CaptureStatusFunc func(ctx context.Context) (CaptureStatusResp, error)
|
||||
|
||||
// EnrollSpeakerFunc / ListSpeakersFunc / ForgetSpeakerFunc — the core-side
|
||||
// halves of voice enrolment.
|
||||
type EnrollSpeakerFunc func(ctx context.Context, req EnrollSpeakerReq) (EnrollSpeakerResp, error)
|
||||
type ListSpeakersFunc func(ctx context.Context) (ListSpeakersResp, error)
|
||||
type ForgetSpeakerFunc func(ctx context.Context, req ForgetSpeakerReq) error
|
||||
|
||||
// CheckFunc — the auth hook signature. Wired by the daemon (auth.Gate.Check
|
||||
// satisfies this); dispatch calls it once per request after param-unmarshal
|
||||
// independence (it gets the raw params, may unmarshal what it needs — ipc
|
||||
@@ -1024,6 +1037,43 @@ func (s *Server) dispatch(ctx context.Context, req Request) (json.RawMessage, er
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodEnrollSpeaker:
|
||||
if s.EnrollSpeakerFn != nil {
|
||||
var p EnrollSpeakerReq
|
||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
resp, err := s.EnrollSpeakerFn(ctx, p)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return marshalResult(resp), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodListSpeakers:
|
||||
if s.ListSpeakersFn != nil {
|
||||
resp, err := s.ListSpeakersFn(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return marshalResult(resp), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodForgetSpeaker:
|
||||
if s.ForgetSpeakerFn != nil {
|
||||
var p ForgetSpeakerReq
|
||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := s.ForgetSpeakerFn(ctx, p); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return marshalResult(nil), nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||
|
||||
case MethodModelStatus:
|
||||
if s.ModelStatusFn != nil {
|
||||
resp, err := s.ModelStatusFn(ctx)
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
package ipc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
)
|
||||
|
||||
// The default that matters most for a biometric: on a core that was never
|
||||
// configured with a speaker block, there is no wire path that takes a
|
||||
// voiceprint, and none that lists the ones that might exist.
|
||||
func TestSpeaker_OffUnlessConfigured(t *testing.T) {
|
||||
_, _, cli, _ := newServerWithStore(t)
|
||||
ctx := context.Background()
|
||||
|
||||
if _, err := cli.EnrollSpeaker(ctx, EnrollSpeakerReq{ID: "kami"}); !errors.Is(err, ErrUnknownMethod) {
|
||||
t.Errorf("EnrollSpeaker error = %v, want ErrUnknownMethod", err)
|
||||
}
|
||||
if _, err := cli.ListSpeakers(ctx); !errors.Is(err, ErrUnknownMethod) {
|
||||
t.Errorf("ListSpeakers error = %v, want ErrUnknownMethod", err)
|
||||
}
|
||||
if err := cli.ForgetSpeaker(ctx, "kami"); !errors.Is(err, ErrUnknownMethod) {
|
||||
t.Errorf("ForgetSpeaker error = %v, want ErrUnknownMethod", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Enrolment carries several samples across the boundary byte for byte — a
|
||||
// profile averaged over the wrong bytes is a profile of nobody.
|
||||
func TestSpeaker_EnrollCrossesTheWire(t *testing.T) {
|
||||
_, srv, cli, _ := newServerWithStore(t)
|
||||
ctx := context.Background()
|
||||
|
||||
enrolled := time.Now().UTC().Truncate(time.Second)
|
||||
var gotID, gotName string
|
||||
var gotSamples [][]byte
|
||||
|
||||
srv.EnrollSpeakerFn = func(_ context.Context, req EnrollSpeakerReq) (EnrollSpeakerResp, error) {
|
||||
gotID, gotName = req.ID, req.Name
|
||||
for _, s := range req.Samples {
|
||||
gotSamples = append(gotSamples, s.Bytes)
|
||||
}
|
||||
return EnrollSpeakerResp{Speaker: Speaker{
|
||||
ID: req.ID, Name: req.Name, Enrolled: enrolled, Samples: len(req.Samples),
|
||||
}}, nil
|
||||
}
|
||||
|
||||
mk := func(b byte, n int) audio.Audio {
|
||||
buf := make([]byte, n)
|
||||
for i := range buf {
|
||||
buf[i] = b
|
||||
}
|
||||
return audio.Audio{Format: audio.PCM16kMono, Bytes: buf}
|
||||
}
|
||||
samples := []audio.Audio{mk(1, 64), mk(2, 96), mk(3, 128)}
|
||||
|
||||
resp, err := cli.EnrollSpeaker(ctx, EnrollSpeakerReq{ID: "kami", Name: "Ками", Samples: samples})
|
||||
if err != nil {
|
||||
t.Fatalf("EnrollSpeaker: %v", err)
|
||||
}
|
||||
if gotID != "kami" || gotName != "Ками" {
|
||||
t.Errorf("server saw id=%q name=%q", gotID, gotName)
|
||||
}
|
||||
if len(gotSamples) != 3 {
|
||||
t.Fatalf("server saw %d samples, want 3", len(gotSamples))
|
||||
}
|
||||
for i, want := range samples {
|
||||
if string(gotSamples[i]) != string(want.Bytes) {
|
||||
t.Errorf("sample %d altered in transit", i)
|
||||
}
|
||||
}
|
||||
if resp.Speaker.Samples != 3 || !resp.Speaker.Enrolled.Equal(enrolled) {
|
||||
t.Errorf("profile came back wrong: %+v", resp.Speaker)
|
||||
}
|
||||
}
|
||||
|
||||
// A listing says who is enrolled and whether recognition actually works. On
|
||||
// this box the honest answer is "enrolled, not recognising", and the response
|
||||
// has to be able to say so — otherwise a surface implies Maven knows who is
|
||||
// talking when nothing on disk can tell.
|
||||
func TestSpeaker_ListReportsDisabledRecognition(t *testing.T) {
|
||||
_, srv, cli, _ := newServerWithStore(t)
|
||||
ctx := context.Background()
|
||||
|
||||
srv.ListSpeakersFn = func(context.Context) (ListSpeakersResp, error) {
|
||||
return ListSpeakersResp{
|
||||
Speakers: []Speaker{{ID: "kami", Name: "Ками", Samples: 3}},
|
||||
Enabled: false,
|
||||
}, nil
|
||||
}
|
||||
|
||||
resp, err := cli.ListSpeakers(ctx)
|
||||
if err != nil {
|
||||
t.Fatalf("ListSpeakers: %v", err)
|
||||
}
|
||||
if len(resp.Speakers) != 1 || resp.Speakers[0].ID != "kami" {
|
||||
t.Fatalf("speakers = %+v", resp.Speakers)
|
||||
}
|
||||
if resp.Enabled {
|
||||
t.Error("Enabled = true; the seam must be able to report that nothing recognises")
|
||||
}
|
||||
}
|
||||
|
||||
// Deletion reaches core with the id intact and reports success. This is the
|
||||
// request that must always work.
|
||||
func TestSpeaker_ForgetReachesCore(t *testing.T) {
|
||||
_, srv, cli, _ := newServerWithStore(t)
|
||||
ctx := context.Background()
|
||||
|
||||
var forgot string
|
||||
srv.ForgetSpeakerFn = func(_ context.Context, req ForgetSpeakerReq) error {
|
||||
forgot = req.ID
|
||||
return nil
|
||||
}
|
||||
if err := cli.ForgetSpeaker(ctx, "гость"); err != nil {
|
||||
t.Fatalf("ForgetSpeaker: %v", err)
|
||||
}
|
||||
if forgot != "гость" {
|
||||
t.Errorf("core forgot %q, want %q", forgot, "гость")
|
||||
}
|
||||
}
|
||||
@@ -59,6 +59,9 @@ const (
|
||||
MethodCaptureAppend Method = "capture_append"
|
||||
MethodCaptureStop Method = "capture_stop"
|
||||
MethodCaptureStatus Method = "capture_status"
|
||||
MethodEnrollSpeaker Method = "enroll_speaker"
|
||||
MethodListSpeakers Method = "list_speakers"
|
||||
MethodForgetSpeaker Method = "forget_speaker"
|
||||
)
|
||||
|
||||
// Request — one frame from module to core. Params is the JSON-encoded argument
|
||||
|
||||
@@ -3,6 +3,7 @@ package memory
|
||||
import (
|
||||
"context"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
)
|
||||
|
||||
@@ -19,6 +20,33 @@ type Store interface {
|
||||
Search(ctx context.Context, vec []float32, topK int) ([]Result, error)
|
||||
}
|
||||
|
||||
// Record is a stored vector read back whole — id, vector and metadata — as
|
||||
// opposed to Result, which is a search hit and carries a score instead of the
|
||||
// vector.
|
||||
type Record struct {
|
||||
ID string
|
||||
Vec []float32
|
||||
Meta map[string]string
|
||||
}
|
||||
|
||||
// Catalog is a Store that can also be enumerated by id prefix and deleted from.
|
||||
//
|
||||
// Search is not enough for every user of the vector table. Speaker profiles
|
||||
// (internal/speaker) need to list exactly their own rows without scoring
|
||||
// anything, because listing enrolled voices is not a similarity question, and
|
||||
// they need Delete because a voiceprint is data about a person and "forget this
|
||||
// voice" has to actually remove it. Note and fact recall use plain Store and are
|
||||
// unaffected.
|
||||
type Catalog interface {
|
||||
Store
|
||||
// ByPrefix returns every row whose id starts with prefix, in no particular
|
||||
// order. An empty prefix returns everything.
|
||||
ByPrefix(ctx context.Context, prefix string) ([]Record, error)
|
||||
// Delete removes one row by id. Deleting a row that is not there is not an
|
||||
// error: the caller asked for it to be gone and it is gone.
|
||||
Delete(ctx context.Context, id string) error
|
||||
}
|
||||
|
||||
// item is a single stored vector with metadata.
|
||||
type item struct {
|
||||
id string
|
||||
@@ -32,14 +60,54 @@ type InMemoryStore struct {
|
||||
items []item
|
||||
}
|
||||
|
||||
// compile-time check: InMemoryStore satisfies Catalog.
|
||||
var _ Catalog = (*InMemoryStore)(nil)
|
||||
|
||||
func NewInMemoryStore() *InMemoryStore {
|
||||
return &InMemoryStore{}
|
||||
}
|
||||
|
||||
// Insert upserts by id, matching the persistent store.MemoryStore: a repeated
|
||||
// id replaces the prior row rather than accumulating a second copy. Re-indexing
|
||||
// a note is an update, and re-enrolling a voice must replace the old voiceprint
|
||||
// rather than leave it searchable.
|
||||
func (s *InMemoryStore) Insert(_ context.Context, id string, vec []float32, meta map[string]string) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for i := range s.items {
|
||||
if s.items[i].id == id {
|
||||
s.items[i] = item{id: id, vec: vec, meta: meta}
|
||||
return nil
|
||||
}
|
||||
}
|
||||
s.items = append(s.items, item{id: id, vec: vec, meta: meta})
|
||||
s.mu.Unlock()
|
||||
return nil
|
||||
}
|
||||
|
||||
// ByPrefix implements Catalog.
|
||||
func (s *InMemoryStore) ByPrefix(_ context.Context, prefix string) ([]Record, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
var out []Record
|
||||
for _, it := range s.items {
|
||||
if !strings.HasPrefix(it.id, prefix) {
|
||||
continue
|
||||
}
|
||||
out = append(out, Record{ID: it.id, Vec: append([]float32(nil), it.vec...), Meta: it.meta})
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Delete implements Catalog.
|
||||
func (s *InMemoryStore) Delete(_ context.Context, id string) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
for i := range s.items {
|
||||
if s.items[i].id == id {
|
||||
s.items = append(s.items[:i], s.items[i+1:]...)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ package memory
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"math"
|
||||
"testing"
|
||||
)
|
||||
@@ -40,8 +41,9 @@ func TestTopKTruncation(t *testing.T) {
|
||||
s := NewInMemoryStore()
|
||||
ctx := context.Background()
|
||||
|
||||
// Distinct ids: Insert upserts by id, so ten rows need ten ids.
|
||||
for i := 0; i < 10; i++ {
|
||||
s.Insert(ctx, "", []float32{float32(i) / 10, 0, 0}, nil)
|
||||
s.Insert(ctx, fmt.Sprintf("n%d", i), []float32{float32(i) / 10, 0, 0}, nil)
|
||||
}
|
||||
|
||||
results, err := s.Search(ctx, []float32{1, 0, 0}, 3)
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
package speaker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
)
|
||||
|
||||
// Enroll registers a voice under an id and a spoken name.
|
||||
//
|
||||
// Several separate samples are required (MinEnrollSamples, MinEnrollSeconds
|
||||
// total): a profile built from one sentence encodes that sentence as much as the
|
||||
// person, and the resulting threshold behaviour is unpredictable. The samples
|
||||
// are embedded individually and the voiceprints averaged, then re-normalised.
|
||||
//
|
||||
// Re-enrolling an existing id REPLACES the profile. That is the intended way to
|
||||
// improve a weak one, and it is why the store upserts by id.
|
||||
//
|
||||
// # The refused step
|
||||
//
|
||||
// The plan document's fourth bullet reads "unknown speakers are enrolled on
|
||||
// first interaction (prompt: 'кто это?')". That is refused. Enrolling a voice is
|
||||
// taking a biometric of a person; doing it automatically the first time someone
|
||||
// walks past the microphone is doing it to guests, without them being part of
|
||||
// the exchange, and a TTS question into a room is not consent from whoever
|
||||
// happens to answer. Enrolment here is an explicit act: an id, a name, and
|
||||
// samples deliberately recorded for the purpose. An unknown voice stays unknown,
|
||||
// which the rest of the system is built to cope with.
|
||||
func (r *Recognizer) Enroll(ctx context.Context, id, name string, samples []audio.Audio) (Profile, error) {
|
||||
id = NormalizeID(id)
|
||||
if !ValidID(id) {
|
||||
return Profile{}, fmt.Errorf("%w: %q", ErrBadID, id)
|
||||
}
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
name = id
|
||||
}
|
||||
if len(samples) < MinEnrollSamples {
|
||||
return Profile{}, fmt.Errorf("%w: %d sample(s), need %d separate ones",
|
||||
ErrTooShort, len(samples), MinEnrollSamples)
|
||||
}
|
||||
|
||||
var total float64
|
||||
for i, s := range samples {
|
||||
if !s.Format.IsValid() {
|
||||
return Profile{}, fmt.Errorf("%w: sample %d: %+v", ErrBadFormat, i+1, s.Format)
|
||||
}
|
||||
total += seconds(s)
|
||||
}
|
||||
if total < MinEnrollSeconds {
|
||||
return Profile{}, fmt.Errorf("%w: %.1fs total, need %.1fs",
|
||||
ErrTooShort, total, MinEnrollSeconds)
|
||||
}
|
||||
|
||||
// Embed first, store second. A model failure halfway through must not leave
|
||||
// a half-built profile that would then be matched against.
|
||||
var (
|
||||
sum []float32
|
||||
dim int
|
||||
)
|
||||
for i, s := range samples {
|
||||
vec, err := r.embed(ctx, s)
|
||||
if err != nil {
|
||||
return Profile{}, fmt.Errorf("speaker: enroll %q sample %d: %w", id, i+1, err)
|
||||
}
|
||||
if sum == nil {
|
||||
sum = make([]float32, len(vec))
|
||||
dim = len(vec)
|
||||
} else if len(vec) != dim {
|
||||
// One model, one width. A mixed-width average would be nonsense.
|
||||
return Profile{}, fmt.Errorf("%w: sample %d is %d wide, expected %d",
|
||||
ErrBadVector, i+1, len(vec), dim)
|
||||
}
|
||||
for j, f := range vec {
|
||||
sum[j] += f
|
||||
}
|
||||
}
|
||||
mean, err := Normalize(sum)
|
||||
if err != nil {
|
||||
// Samples that cancel each other out to zero are not one voice.
|
||||
return Profile{}, fmt.Errorf("speaker: enroll %q: %w", id, err)
|
||||
}
|
||||
|
||||
p := Profile{
|
||||
ID: id,
|
||||
Name: name,
|
||||
Enrolled: r.now().UTC(),
|
||||
Samples: len(samples),
|
||||
Dim: dim,
|
||||
Vec: mean,
|
||||
}
|
||||
meta := map[string]string{
|
||||
"name": p.Name,
|
||||
"samples": strconv.Itoa(p.Samples),
|
||||
"enrolled": p.Enrolled.Format(time.RFC3339),
|
||||
// kind marks the row for anything walking the vector table, so a future
|
||||
// export or debug page can tell a voiceprint from a note embedding
|
||||
// without parsing the id.
|
||||
"kind": "speaker",
|
||||
}
|
||||
if err := r.cat.Insert(ctx, Prefix+id, mean, meta); err != nil {
|
||||
return Profile{}, fmt.Errorf("speaker: enroll %q: %w", id, err)
|
||||
}
|
||||
return p, nil
|
||||
}
|
||||
@@ -0,0 +1,212 @@
|
||||
package speaker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"sort"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
)
|
||||
|
||||
// Recognizer holds the embedder and the enrolled profiles.
|
||||
//
|
||||
// The profiles live in the shared vector table under the "speaker:" id prefix,
|
||||
// which is what the plan asked for and what keeps them inside the encrypted
|
||||
// store rather than in a sidecar file. They are read through memory.Catalog
|
||||
// (ByPrefix / Delete) rather than Search, because "who is enrolled" is not a
|
||||
// similarity question and note recall must never rank a voiceprint.
|
||||
type Recognizer struct {
|
||||
emb Embedder
|
||||
cat memory.Catalog
|
||||
threshold float64
|
||||
minSec float64
|
||||
now func() time.Time
|
||||
}
|
||||
|
||||
// Config — the recognizer's knobs, built from config.SpeakerConfig.
|
||||
type Config struct {
|
||||
// Threshold — cosine similarity a match must beat. 0 ⇒ DefaultThreshold.
|
||||
Threshold float64
|
||||
// MinSeconds — least speech an identification will look at. 0 ⇒
|
||||
// DefaultMinSeconds.
|
||||
MinSeconds float64
|
||||
}
|
||||
|
||||
// New builds a Recognizer. emb nil ⇒ Disabled, which is this box's state and
|
||||
// makes every Identify answer ErrDisabled while enrolment and listing still
|
||||
// behave sensibly (they refuse for the same reason, with the same error).
|
||||
func New(emb Embedder, cat memory.Catalog, cfg Config) (*Recognizer, error) {
|
||||
if cat == nil {
|
||||
return nil, fmt.Errorf("speaker: no profile store")
|
||||
}
|
||||
if emb == nil {
|
||||
emb = Disabled{}
|
||||
}
|
||||
th := cfg.Threshold
|
||||
if th <= 0 {
|
||||
th = DefaultThreshold
|
||||
}
|
||||
min := cfg.MinSeconds
|
||||
if min <= 0 {
|
||||
min = DefaultMinSeconds
|
||||
}
|
||||
return &Recognizer{emb: emb, cat: cat, threshold: th, minSec: min, now: time.Now}, nil
|
||||
}
|
||||
|
||||
// Enabled reports whether an embedding model is actually wired. Surfaces use it
|
||||
// to say "recognition is off" once instead of failing every turn.
|
||||
func (r *Recognizer) Enabled() bool {
|
||||
_, disabled := r.emb.(Disabled)
|
||||
return !disabled
|
||||
}
|
||||
|
||||
// Threshold is the configured match floor, for a status line.
|
||||
func (r *Recognizer) Threshold() float64 { return r.threshold }
|
||||
|
||||
// Identify names the voice in a. ErrUnknown when nothing is close enough, which
|
||||
// is a normal answer and not a failure: a guest is a guest, and the caller
|
||||
// carries on with no speaker attached rather than guessing.
|
||||
//
|
||||
// Identification never decides whether Maven listens. It annotates the turn.
|
||||
func (r *Recognizer) Identify(ctx context.Context, a audio.Audio) (Match, error) {
|
||||
if !a.Format.IsValid() {
|
||||
return Match{}, fmt.Errorf("%w: %+v", ErrBadFormat, a.Format)
|
||||
}
|
||||
if seconds(a) < r.minSec {
|
||||
return Match{}, fmt.Errorf("%w: %.1fs, need %.1fs", ErrTooShort, seconds(a), r.minSec)
|
||||
}
|
||||
vec, err := r.embed(ctx, a)
|
||||
if err != nil {
|
||||
return Match{}, err
|
||||
}
|
||||
profiles, err := r.List(ctx)
|
||||
if err != nil {
|
||||
return Match{}, err
|
||||
}
|
||||
if len(profiles) == 0 {
|
||||
return Match{}, ErrNoProfiles
|
||||
}
|
||||
|
||||
best := Match{Score: -2}
|
||||
for _, p := range profiles {
|
||||
if s := Similarity(vec, p.Vec); s > best.Score {
|
||||
best = Match{Profile: p, Score: s}
|
||||
}
|
||||
}
|
||||
if best.Score < r.threshold {
|
||||
// The closest profile is reported in the error for a log line, because
|
||||
// "не узнала, ближе всего Ками на 0.61" is what makes a threshold
|
||||
// tunable. The caller must not use it as an identification.
|
||||
return Match{}, fmt.Errorf("%w (closest %s at %.2f, need %.2f)",
|
||||
ErrUnknown, best.Profile.ID, best.Score, r.threshold)
|
||||
}
|
||||
return best, nil
|
||||
}
|
||||
|
||||
// List returns every enrolled profile, sorted by id so a listing is stable.
|
||||
func (r *Recognizer) List(ctx context.Context) ([]Profile, error) {
|
||||
recs, err := r.cat.ByPrefix(ctx, Prefix)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("speaker: list: %w", err)
|
||||
}
|
||||
out := make([]Profile, 0, len(recs))
|
||||
for _, rec := range recs {
|
||||
out = append(out, profileFromRecord(rec))
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].ID < out[j].ID })
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Get returns one profile by id.
|
||||
func (r *Recognizer) Get(ctx context.Context, id string) (Profile, error) {
|
||||
id = NormalizeID(id)
|
||||
if !ValidID(id) {
|
||||
return Profile{}, fmt.Errorf("%w: %q", ErrBadID, id)
|
||||
}
|
||||
recs, err := r.cat.ByPrefix(ctx, Prefix+id)
|
||||
if err != nil {
|
||||
return Profile{}, fmt.Errorf("speaker: get: %w", err)
|
||||
}
|
||||
for _, rec := range recs {
|
||||
if rec.ID == Prefix+id {
|
||||
return profileFromRecord(rec), nil
|
||||
}
|
||||
}
|
||||
return Profile{}, fmt.Errorf("%w: %q", ErrNotFound, id)
|
||||
}
|
||||
|
||||
// Forget deletes a profile. This is the one operation that must always work:
|
||||
// a voiceprint is data about a person, and "перестань узнавать её" has to
|
||||
// actually remove it, not mark it inactive.
|
||||
func (r *Recognizer) Forget(ctx context.Context, id string) error {
|
||||
id = NormalizeID(id)
|
||||
if !ValidID(id) {
|
||||
return fmt.Errorf("%w: %q", ErrBadID, id)
|
||||
}
|
||||
if _, err := r.Get(ctx, id); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := r.cat.Delete(ctx, Prefix+id); err != nil {
|
||||
return fmt.Errorf("speaker: forget %q: %w", id, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// embed runs the model and normalises the result.
|
||||
func (r *Recognizer) embed(ctx context.Context, a audio.Audio) ([]float32, error) {
|
||||
raw, err := r.emb.Embed(ctx, a)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
vec, err := Normalize(raw)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return vec, nil
|
||||
}
|
||||
|
||||
// profileFromRecord reads a stored row back into a Profile. A row with
|
||||
// unreadable metadata still yields a usable voiceprint — the vector is the part
|
||||
// that matters, and losing a name should not lose the enrolment.
|
||||
func profileFromRecord(rec memory.Record) Profile {
|
||||
p := Profile{
|
||||
ID: trimPrefix(rec.ID),
|
||||
Vec: rec.Vec,
|
||||
Dim: len(rec.Vec),
|
||||
Name: rec.Meta["name"],
|
||||
}
|
||||
if s := rec.Meta["samples"]; s != "" {
|
||||
p.Samples = atoi(s)
|
||||
}
|
||||
if ts := rec.Meta["enrolled"]; ts != "" {
|
||||
if t, err := time.Parse(time.RFC3339, ts); err == nil {
|
||||
p.Enrolled = t
|
||||
}
|
||||
}
|
||||
if p.Name == "" {
|
||||
p.Name = p.ID
|
||||
}
|
||||
return p
|
||||
}
|
||||
|
||||
func trimPrefix(id string) string {
|
||||
if len(id) > len(Prefix) && id[:len(Prefix)] == Prefix {
|
||||
return id[len(Prefix):]
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
// atoi is a tolerant small-integer parse: metadata that is not a number reads
|
||||
// as 0 rather than failing the whole listing.
|
||||
func atoi(s string) int {
|
||||
n := 0
|
||||
for _, r := range s {
|
||||
if r < '0' || r > '9' {
|
||||
return 0
|
||||
}
|
||||
n = n*10 + int(r-'0')
|
||||
}
|
||||
return n
|
||||
}
|
||||
@@ -0,0 +1,223 @@
|
||||
// Package speaker is voice identification (Vikunja #255,
|
||||
// docs/plans/10-speaker-recognition.md).
|
||||
//
|
||||
// The shape is the same seam internal/vision uses: an Embedder turns audio into
|
||||
// a voiceprint, a Recognizer compares one against the enrolled profiles, and a
|
||||
// Disabled floor refuses politely when nothing is wired. On this box nothing is
|
||||
// wired, and that is the honest state — see "Blocked" below.
|
||||
//
|
||||
// # A voiceprint is not like the other vectors
|
||||
//
|
||||
// Everything else in the vector table is something he wrote or said. A speaker
|
||||
// profile is biometric data about a person, quite possibly a person who never
|
||||
// asked for Maven to exist. The rules that follow from that are in the code:
|
||||
//
|
||||
// - Enrolment is explicit and named. There is no "enrol the unknown voice
|
||||
// automatically" path; see the refusal in enroll.go.
|
||||
// - A profile is deletable, individually, and Forget really removes the row.
|
||||
// - Below the threshold the answer is "I do not know", never the closest
|
||||
// guess. A misattributed fact is worse than an unattributed one.
|
||||
// - Nothing here gates whether Maven listens or answers. Identification
|
||||
// annotates a turn; it never authorises one, and an unrecognised voice is
|
||||
// not turned away.
|
||||
// - Voiceprints never leave the box. They live in the encrypted store with
|
||||
// everything else and are never search input to anything external.
|
||||
//
|
||||
// # Blocked
|
||||
//
|
||||
// There is no speaker-embedding model on this box: no ECAPA-TDNN, no x-vector,
|
||||
// no wespeaker or titanet ONNX anywhere under /mnt/hdd1 or models/ (checked
|
||||
// 2026-08-01; the only ONNX files are the e5 text embedder and the piper voice).
|
||||
// There are also no enrolment samples. So Recognizer runs against Disabled and
|
||||
// every Identify answers ErrDisabled until a model lands.
|
||||
//
|
||||
// The MFCC + GMM "simplest floor" in the plan document is refused rather than
|
||||
// deferred. A hand-rolled spectral distance would identify people confidently
|
||||
// and wrongly, and its output would be written into facts as "Ками said this".
|
||||
// For a biometric, a bad floor is worse than none: no answer is honest, and a
|
||||
// wrong answer is a false memory about a person.
|
||||
package speaker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"math"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
)
|
||||
|
||||
// Prefix — the id prefix speaker profiles carry in the shared vector table.
|
||||
// It is what ByPrefix enumerates and what keeps voiceprints out of note recall.
|
||||
const Prefix = "speaker:"
|
||||
|
||||
// DefaultThreshold — cosine similarity a match must beat to be a match.
|
||||
//
|
||||
// 0.7 is the usual operating point for ECAPA-style embeddings on clean speech
|
||||
// and it is deliberately on the strict side here. The two error directions are
|
||||
// not symmetric: refusing to name a voice costs a "не узнала", while naming the
|
||||
// wrong person writes his wife's remark into a fact attributed to him.
|
||||
const DefaultThreshold = 0.7
|
||||
|
||||
// DefaultMinSeconds — how much speech an identification needs. Under about two
|
||||
// seconds a voiceprint is mostly noise and the similarity score is not worth
|
||||
// reading.
|
||||
const DefaultMinSeconds = 2.0
|
||||
|
||||
// MinEnrollSamples / MinEnrollSeconds — what enrolment requires. Several
|
||||
// separate utterances, not one long one: a profile built from a single sentence
|
||||
// encodes that sentence's prosody as much as the voice.
|
||||
const (
|
||||
MinEnrollSamples = 3
|
||||
MinEnrollSeconds = 9.0
|
||||
)
|
||||
|
||||
// Errors callers distinguish.
|
||||
var (
|
||||
// ErrDisabled — no embedding model is wired. The state of this box.
|
||||
ErrDisabled = errors.New("speaker: recognition is not configured")
|
||||
// ErrTooShort — not enough speech to say anything about.
|
||||
ErrTooShort = errors.New("speaker: not enough audio")
|
||||
// ErrUnknown — audio embedded fine, but no enrolled profile is close
|
||||
// enough. Not an error in the sense of something being broken: it is the
|
||||
// correct answer for a guest, and the caller should carry on without a
|
||||
// speaker rather than treat the turn as failed.
|
||||
ErrUnknown = errors.New("speaker: voice not recognised")
|
||||
// ErrNoProfiles — nobody is enrolled yet.
|
||||
ErrNoProfiles = errors.New("speaker: nobody is enrolled")
|
||||
// ErrNotFound — no profile with that id.
|
||||
ErrNotFound = errors.New("speaker: no such profile")
|
||||
// ErrBadID — an id that is empty or carries characters an id should not.
|
||||
ErrBadID = errors.New("speaker: invalid profile id")
|
||||
// ErrBadFormat — audio that is not the canonical 16 kHz mono PCM shape.
|
||||
ErrBadFormat = errors.New("speaker: audio format not supported")
|
||||
// ErrBadVector — an embedder returned something unusable (empty, or all
|
||||
// zeroes, which normalises to nothing and would match everything equally).
|
||||
ErrBadVector = errors.New("speaker: embedder returned an unusable vector")
|
||||
)
|
||||
|
||||
// Embedder turns speech into a voiceprint. Implementations are expected to
|
||||
// return an L2-normalised vector, because the whole store compares by dot
|
||||
// product; Normalize is applied anyway rather than trusted.
|
||||
//
|
||||
// This is the seam a downloaded ECAPA-TDNN ONNX model plugs into. It is an
|
||||
// interface rather than a concrete ONNX type so the package is testable with no
|
||||
// model on disk, which is the only way it could be tested here at all.
|
||||
type Embedder interface {
|
||||
Embed(ctx context.Context, a audio.Audio) ([]float32, error)
|
||||
// Dim is the vector width, used to reject a profile recorded with a
|
||||
// different model rather than silently scoring it as zero.
|
||||
Dim() int
|
||||
}
|
||||
|
||||
// Disabled is the floor: no model, no answers, no guesses.
|
||||
type Disabled struct{}
|
||||
|
||||
// Embed always fails with ErrDisabled.
|
||||
func (Disabled) Embed(context.Context, audio.Audio) ([]float32, error) { return nil, ErrDisabled }
|
||||
|
||||
// Dim is 0 for the disabled embedder.
|
||||
func (Disabled) Dim() int { return 0 }
|
||||
|
||||
// Profile — one enrolled voice.
|
||||
//
|
||||
// Name is what she calls the person out loud ("Ками"). ID is the stable handle
|
||||
// used in sources and metadata. Samples records how many utterances the
|
||||
// voiceprint was averaged from, so a profile enrolled from the bare minimum is
|
||||
// visibly weaker than one built from ten.
|
||||
type Profile struct {
|
||||
ID string `json:"id"`
|
||||
Name string `json:"name"`
|
||||
Enrolled time.Time `json:"enrolled"`
|
||||
Samples int `json:"samples"`
|
||||
Dim int `json:"dim"`
|
||||
|
||||
// Vec is the voiceprint. Not serialised to any surface: a listing tells him
|
||||
// who is enrolled, it does not hand out the biometric itself.
|
||||
Vec []float32 `json:"-"`
|
||||
}
|
||||
|
||||
// Source is what a fact or note written during this speaker's turn is tagged
|
||||
// with, e.g. "tap:voice:speaker:kami". Attribution belongs in the source rather
|
||||
// than in the text, so it can be corrected or dropped later without rewriting
|
||||
// what was said.
|
||||
func (p Profile) Source(base string) string {
|
||||
if p.ID == "" {
|
||||
return base
|
||||
}
|
||||
return base + ":" + Prefix + p.ID
|
||||
}
|
||||
|
||||
// Match — an identification result. Score is cosine similarity in [-1, 1].
|
||||
type Match struct {
|
||||
Profile Profile
|
||||
Score float64
|
||||
}
|
||||
|
||||
// ValidID reports whether an id is usable as a profile handle. Deliberately
|
||||
// narrow: lowercase letters, digits, dash and underscore. Ids end up in note
|
||||
// sources and in vector-table keys, so a permissive id would be a way to write
|
||||
// into a neighbouring key space.
|
||||
func ValidID(id string) bool {
|
||||
if id == "" || len(id) > 64 {
|
||||
return false
|
||||
}
|
||||
for _, r := range id {
|
||||
switch {
|
||||
case r >= 'a' && r <= 'z', r >= '0' && r <= '9', r == '-', r == '_':
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// NormalizeID lowercases and trims a proposed id before validating it, so
|
||||
// "Ками" typed as "Kami " does not fail for a reason nobody can see.
|
||||
func NormalizeID(id string) string {
|
||||
return strings.ToLower(strings.TrimSpace(id))
|
||||
}
|
||||
|
||||
// Normalize returns an L2-normalised copy of v, or ErrBadVector when there is
|
||||
// nothing to normalise. A zero vector is refused rather than passed on: it
|
||||
// scores 0 against everything, which reads as "no match" but for the wrong
|
||||
// reason and would hide a broken embedder.
|
||||
func Normalize(v []float32) ([]float32, error) {
|
||||
if len(v) == 0 {
|
||||
return nil, ErrBadVector
|
||||
}
|
||||
var sum float64
|
||||
for _, f := range v {
|
||||
if math.IsNaN(float64(f)) || math.IsInf(float64(f), 0) {
|
||||
return nil, ErrBadVector
|
||||
}
|
||||
sum += float64(f) * float64(f)
|
||||
}
|
||||
norm := math.Sqrt(sum)
|
||||
if norm == 0 {
|
||||
return nil, ErrBadVector
|
||||
}
|
||||
out := make([]float32, len(v))
|
||||
for i, f := range v {
|
||||
out[i] = float32(float64(f) / norm)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Similarity is the cosine similarity of two L2-normalised vectors. Different
|
||||
// widths score 0: a profile enrolled with another model must not accidentally
|
||||
// match, and 0 is below every sane threshold.
|
||||
func Similarity(a, b []float32) float64 {
|
||||
if len(a) != len(b) || len(a) == 0 {
|
||||
return 0
|
||||
}
|
||||
var sum float64
|
||||
for i := range a {
|
||||
sum += float64(a[i]) * float64(b[i])
|
||||
}
|
||||
return sum
|
||||
}
|
||||
|
||||
// seconds is the playback length of a frame, for the minimum-audio checks.
|
||||
func seconds(a audio.Audio) float64 { return a.Duration() }
|
||||
@@ -0,0 +1,397 @@
|
||||
package speaker
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"math"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
"github.com/kami/maven/internal/memory"
|
||||
)
|
||||
|
||||
// fakeEmbedder returns a fixed vector per "voice", so a test can enrol one
|
||||
// person and present another without a model. Wobble adds a small perturbation
|
||||
// so repeated samples of one voice are close but not identical, which is what a
|
||||
// real embedder produces.
|
||||
type fakeEmbedder struct {
|
||||
vec []float32
|
||||
err error
|
||||
calls int
|
||||
wobble float32
|
||||
}
|
||||
|
||||
func (f *fakeEmbedder) Embed(_ context.Context, _ audio.Audio) ([]float32, error) {
|
||||
f.calls++
|
||||
if f.err != nil {
|
||||
return nil, f.err
|
||||
}
|
||||
out := append([]float32(nil), f.vec...)
|
||||
if f.wobble != 0 && len(out) > 1 {
|
||||
out[0] += f.wobble * float32(f.calls)
|
||||
out[1] -= f.wobble * float32(f.calls)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (f *fakeEmbedder) Dim() int { return len(f.vec) }
|
||||
|
||||
// speech builds n seconds of the canonical audio shape.
|
||||
func speech(sec float64) audio.Audio {
|
||||
return audio.Audio{Format: audio.PCM16kMono, Bytes: make([]byte, int(sec*16000)*2)}
|
||||
}
|
||||
|
||||
func newRec(t *testing.T, emb Embedder) (*Recognizer, memory.Catalog) {
|
||||
t.Helper()
|
||||
cat := memory.NewInMemoryStore()
|
||||
r, err := New(emb, cat, Config{})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return r, cat
|
||||
}
|
||||
|
||||
func enrolSamples(n int, sec float64) []audio.Audio {
|
||||
out := make([]audio.Audio, n)
|
||||
for i := range out {
|
||||
out[i] = speech(sec)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// The state of this box: no model on disk. Every identification refuses rather
|
||||
// than guessing, and it says why.
|
||||
func TestDisabledRefusesEverything(t *testing.T) {
|
||||
r, _ := newRec(t, nil)
|
||||
if r.Enabled() {
|
||||
t.Error("a recognizer with no model reports itself enabled")
|
||||
}
|
||||
if _, err := r.Identify(context.Background(), speech(5)); !errors.Is(err, ErrDisabled) {
|
||||
t.Errorf("Identify = %v, want ErrDisabled", err)
|
||||
}
|
||||
if _, err := r.Enroll(context.Background(), "kami", "Ками", enrolSamples(3, 4)); !errors.Is(err, ErrDisabled) {
|
||||
t.Errorf("Enroll = %v, want ErrDisabled", err)
|
||||
}
|
||||
// Listing still works: knowing that nobody is enrolled needs no model.
|
||||
got, err := r.List(context.Background())
|
||||
if err != nil || len(got) != 0 {
|
||||
t.Errorf("List = %v, %v", got, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRequiresAProfileStore(t *testing.T) {
|
||||
if _, err := New(nil, nil, Config{}); err == nil {
|
||||
t.Error("built a recognizer with nowhere to keep profiles")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnrollThenIdentify(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0, 0, 0}, wobble: 0.01}
|
||||
r, _ := newRec(t, emb)
|
||||
ctx := context.Background()
|
||||
|
||||
p, err := r.Enroll(ctx, "Kami ", "Ками", enrolSamples(3, 4))
|
||||
if err != nil {
|
||||
t.Fatalf("enroll: %v", err)
|
||||
}
|
||||
if p.ID != "kami" {
|
||||
t.Errorf("id = %q, want the normalised %q", p.ID, "kami")
|
||||
}
|
||||
if p.Name != "Ками" || p.Samples != 3 || p.Dim != 4 {
|
||||
t.Errorf("profile = %+v", p)
|
||||
}
|
||||
|
||||
m, err := r.Identify(ctx, speech(5))
|
||||
if err != nil {
|
||||
t.Fatalf("identify: %v", err)
|
||||
}
|
||||
if m.Profile.ID != "kami" || m.Profile.Name != "Ками" {
|
||||
t.Errorf("match = %+v", m)
|
||||
}
|
||||
if m.Score < r.Threshold() {
|
||||
t.Errorf("score %.3f is below the threshold it supposedly passed", m.Score)
|
||||
}
|
||||
}
|
||||
|
||||
// The error direction that matters. Naming the wrong person writes a false
|
||||
// memory about them, so a voice that is not close enough gets no name at all.
|
||||
func TestUnfamiliarVoiceIsNotGuessed(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0, 0, 0}}
|
||||
r, _ := newRec(t, emb)
|
||||
ctx := context.Background()
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// A different voice: orthogonal voiceprint, similarity 0.
|
||||
emb.vec = []float32{0, 1, 0, 0}
|
||||
m, err := r.Identify(ctx, speech(5))
|
||||
if !errors.Is(err, ErrUnknown) {
|
||||
t.Fatalf("Identify = %v, want ErrUnknown", err)
|
||||
}
|
||||
if m.Profile.ID != "" {
|
||||
t.Errorf("a refused identification still handed back %q", m.Profile.ID)
|
||||
}
|
||||
// The log line needs the near miss to make the threshold tunable.
|
||||
if !contains(err.Error(), "kami") {
|
||||
t.Errorf("error does not name the closest profile: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Just under the threshold is still unknown. A boundary this important gets its
|
||||
// own test rather than being implied.
|
||||
func TestThresholdIsAFloorNotASuggestion(t *testing.T) {
|
||||
cat := memory.NewInMemoryStore()
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}}
|
||||
r, err := New(emb, cat, Config{Threshold: 0.9})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ctx := context.Background()
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// cos ≈ 0.866, comfortably similar and still not similar enough.
|
||||
emb.vec = []float32{0.866, 0.5}
|
||||
if _, err := r.Identify(ctx, speech(5)); !errors.Is(err, ErrUnknown) {
|
||||
t.Fatalf("0.866 against a 0.9 threshold = %v, want ErrUnknown", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestShortAudioIsRefusedBeforeTheModelRuns(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}}
|
||||
r, _ := newRec(t, emb)
|
||||
if _, err := r.Identify(context.Background(), speech(0.5)); !errors.Is(err, ErrTooShort) {
|
||||
t.Fatalf("got %v, want ErrTooShort", err)
|
||||
}
|
||||
if emb.calls != 0 {
|
||||
t.Error("a half-second of audio was sent to the model anyway")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrongAudioFormatIsRefused(t *testing.T) {
|
||||
r, _ := newRec(t, &fakeEmbedder{vec: []float32{1, 0}})
|
||||
bad := audio.Audio{
|
||||
Format: audio.Format{SampleRate: 44100, Channels: 2, SampleBits: 16, Encoding: "pcm_s16le"},
|
||||
Bytes: make([]byte, 44100*4*5),
|
||||
}
|
||||
if _, err := r.Identify(context.Background(), bad); !errors.Is(err, ErrBadFormat) {
|
||||
t.Fatalf("got %v, want ErrBadFormat", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIdentifyWithNobodyEnrolled(t *testing.T) {
|
||||
r, _ := newRec(t, &fakeEmbedder{vec: []float32{1, 0}})
|
||||
if _, err := r.Identify(context.Background(), speech(5)); !errors.Is(err, ErrNoProfiles) {
|
||||
t.Fatalf("got %v, want ErrNoProfiles", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Enrolment is an explicit act with real samples behind it, not a byproduct of
|
||||
// someone speaking once.
|
||||
func TestEnrollmentRequiresSeveralRealSamples(t *testing.T) {
|
||||
r, _ := newRec(t, &fakeEmbedder{vec: []float32{1, 0}})
|
||||
ctx := context.Background()
|
||||
cases := []struct {
|
||||
name string
|
||||
samples []audio.Audio
|
||||
}{
|
||||
{"one long sample", enrolSamples(1, 30)},
|
||||
{"two samples", enrolSamples(2, 10)},
|
||||
{"three samples but seconds of audio", enrolSamples(3, 1)},
|
||||
{"none at all", nil},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", c.samples); !errors.Is(err, ErrTooShort) {
|
||||
t.Errorf("%s: %v, want ErrTooShort", c.name, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnrollRejectsBadIDs(t *testing.T) {
|
||||
r, _ := newRec(t, &fakeEmbedder{vec: []float32{1, 0}})
|
||||
for _, id := range []string{"", " ", "../etc/passwd", "speaker:kami", "имя", "a/b", "x y"} {
|
||||
if _, err := r.Enroll(context.Background(), id, "n", enrolSamples(3, 4)); !errors.Is(err, ErrBadID) {
|
||||
t.Errorf("id %q accepted or wrong error: %v", id, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Re-enrolling replaces the voiceprint. Leaving the old one searchable would
|
||||
// mean a person's rejected profile keeps matching them.
|
||||
func TestReEnrollReplaces(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0, 0}}
|
||||
r, _ := newRec(t, emb)
|
||||
ctx := context.Background()
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
emb.vec = []float32{0, 1, 0}
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(4, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
list, err := r.List(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(list) != 1 {
|
||||
t.Fatalf("%d profiles after re-enrolling one person", len(list))
|
||||
}
|
||||
if list[0].Samples != 4 {
|
||||
t.Errorf("sample count = %d, want the new 4", list[0].Samples)
|
||||
}
|
||||
// The new voiceprint is the one that matches.
|
||||
if m, err := r.Identify(ctx, speech(5)); err != nil || m.Score < 0.99 {
|
||||
t.Errorf("identify after re-enrol: %v (score %.3f)", err, m.Score)
|
||||
}
|
||||
}
|
||||
|
||||
// "Перестань узнавать её" has to actually delete the biometric.
|
||||
func TestForgetRemovesTheVoiceprint(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}}
|
||||
r, cat := newRec(t, emb)
|
||||
ctx := context.Background()
|
||||
if _, err := r.Enroll(ctx, "guest", "Гостья", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := r.Forget(ctx, "Guest "); err != nil {
|
||||
t.Fatalf("forget: %v", err)
|
||||
}
|
||||
recs, err := cat.ByPrefix(ctx, Prefix)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(recs) != 0 {
|
||||
t.Errorf("%d row(s) survived Forget", len(recs))
|
||||
}
|
||||
if _, err := r.Get(ctx, "guest"); !errors.Is(err, ErrNotFound) {
|
||||
t.Errorf("Get after Forget = %v, want ErrNotFound", err)
|
||||
}
|
||||
if err := r.Forget(ctx, "guest"); !errors.Is(err, ErrNotFound) {
|
||||
t.Errorf("second Forget = %v, want ErrNotFound", err)
|
||||
}
|
||||
}
|
||||
|
||||
// Voiceprints share the vector table with note and fact embeddings, so the
|
||||
// prefix has to actually partition it.
|
||||
func TestProfilesDoNotCollideWithNoteVectors(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}}
|
||||
r, cat := newRec(t, emb)
|
||||
ctx := context.Background()
|
||||
if err := cat.Insert(ctx, "note:1", []float32{1, 0}, map[string]string{"text": "заметка"}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
list, err := r.List(ctx)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(list) != 1 || list[0].ID != "kami" {
|
||||
t.Errorf("listing picked up a non-speaker row: %+v", list)
|
||||
}
|
||||
// And an identical note vector is never returned as a match.
|
||||
m, err := r.Identify(ctx, speech(5))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if m.Profile.ID != "kami" {
|
||||
t.Errorf("matched %q", m.Profile.ID)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmbedderFailurePropagates(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}, err: errors.New("onnx fell over")}
|
||||
r, _ := newRec(t, emb)
|
||||
if _, err := r.Identify(context.Background(), speech(5)); err == nil {
|
||||
t.Error("a model failure was reported as a successful identification")
|
||||
}
|
||||
if _, err := r.Enroll(context.Background(), "kami", "К", enrolSamples(3, 4)); err == nil {
|
||||
t.Error("a model failure produced a profile")
|
||||
}
|
||||
}
|
||||
|
||||
// A zero vector scores 0 against everything, which reads as "no match" for the
|
||||
// wrong reason and would hide a broken model.
|
||||
func TestUnusableVectorsAreRefused(t *testing.T) {
|
||||
r, _ := newRec(t, &fakeEmbedder{vec: []float32{0, 0, 0}})
|
||||
if _, err := r.Enroll(context.Background(), "kami", "К", enrolSamples(3, 4)); !errors.Is(err, ErrBadVector) {
|
||||
t.Errorf("zero vector: %v, want ErrBadVector", err)
|
||||
}
|
||||
if _, err := Normalize(nil); !errors.Is(err, ErrBadVector) {
|
||||
t.Errorf("empty: %v", err)
|
||||
}
|
||||
if _, err := Normalize([]float32{float32(nan())}); !errors.Is(err, ErrBadVector) {
|
||||
t.Errorf("NaN: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeProducesAUnitVector(t *testing.T) {
|
||||
v, err := Normalize([]float32{3, 4})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := Similarity(v, v); got < 0.999 || got > 1.001 {
|
||||
t.Errorf("self-similarity = %f, want 1", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A profile enrolled with another model must not accidentally match.
|
||||
func TestDifferentWidthsScoreZero(t *testing.T) {
|
||||
if got := Similarity([]float32{1, 0}, []float32{1, 0, 0}); got != 0 {
|
||||
t.Errorf("mismatched widths scored %f", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Attribution belongs in the source, so it can be corrected without rewriting
|
||||
// what was said.
|
||||
func TestProfileSource(t *testing.T) {
|
||||
p := Profile{ID: "kami"}
|
||||
if got := p.Source("tap:voice"); got != "tap:voice:speaker:kami" {
|
||||
t.Errorf("source = %q", got)
|
||||
}
|
||||
var anon Profile
|
||||
if got := anon.Source("tap:voice"); got != "tap:voice" {
|
||||
t.Errorf("unattributed source = %q, want the base unchanged", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidID(t *testing.T) {
|
||||
for _, ok := range []string{"kami", "guest-2", "a_b", "x"} {
|
||||
if !ValidID(ok) {
|
||||
t.Errorf("%q rejected", ok)
|
||||
}
|
||||
}
|
||||
for _, bad := range []string{"", "Kami", "имя", "a b", "a/b", "a:b", "..", strings.Repeat("a", 65)} {
|
||||
if ValidID(bad) {
|
||||
t.Errorf("%q accepted", bad)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestProfileMetadataSurvivesARoundTrip(t *testing.T) {
|
||||
emb := &fakeEmbedder{vec: []float32{1, 0}}
|
||||
r, _ := newRec(t, emb)
|
||||
r.now = func() time.Time { return time.Date(2026, 8, 1, 12, 0, 0, 0, time.UTC) }
|
||||
ctx := context.Background()
|
||||
if _, err := r.Enroll(ctx, "kami", "Ками", enrolSamples(3, 4)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
got, err := r.Get(ctx, "kami")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got.Name != "Ками" || got.Samples != 3 {
|
||||
t.Errorf("profile = %+v", got)
|
||||
}
|
||||
if !got.Enrolled.Equal(time.Date(2026, 8, 1, 12, 0, 0, 0, time.UTC)) {
|
||||
t.Errorf("enrolled = %v", got.Enrolled)
|
||||
}
|
||||
}
|
||||
|
||||
func contains(s, sub string) bool { return strings.Contains(s, sub) }
|
||||
|
||||
func nan() float64 { return math.NaN() }
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"fmt"
|
||||
"math"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/memory"
|
||||
@@ -37,8 +38,10 @@ func (s *Store) VectorMemory() *MemoryStore {
|
||||
return &MemoryStore{db: s.db}
|
||||
}
|
||||
|
||||
// compile-time check: MemoryStore satisfies the memory.Store interface.
|
||||
// compile-time check: MemoryStore satisfies the memory.Store interface, and the
|
||||
// wider Catalog that speaker profiles need (enumerate by prefix, delete by id).
|
||||
var _ memory.Store = (*MemoryStore)(nil)
|
||||
var _ memory.Catalog = (*MemoryStore)(nil)
|
||||
|
||||
// Insert upserts a vector by id: a repeated id replaces the prior row rather
|
||||
// than accumulating duplicates (the note/fact ids are stable and unique, so a
|
||||
@@ -95,6 +98,56 @@ func (m *MemoryStore) Search(ctx context.Context, vec []float32, topK int) ([]me
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// ByPrefix returns every row whose id starts with prefix, vectors included.
|
||||
//
|
||||
// This is not a similarity query and deliberately does not score anything:
|
||||
// listing the enrolled voices is a question about which rows exist, and asking
|
||||
// it through Search would mean inventing a query vector to rank them by. The
|
||||
// prefix is matched with LIKE against an escaped pattern, so a profile id
|
||||
// containing % or _ cannot widen the match.
|
||||
func (m *MemoryStore) ByPrefix(ctx context.Context, prefix string) ([]memory.Record, error) {
|
||||
pattern := escapeLike(prefix) + "%"
|
||||
rows, err := m.db.QueryContext(ctx,
|
||||
`SELECT id, vec, meta FROM memory_vectors WHERE id LIKE ? ESCAPE '\'`, pattern)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("memory: by prefix %q: %w", prefix, err)
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var out []memory.Record
|
||||
for rows.Next() {
|
||||
var id, metaJSON string
|
||||
var blob []byte
|
||||
if err := rows.Scan(&id, &blob, &metaJSON); err != nil {
|
||||
return nil, fmt.Errorf("memory: row: %w", err)
|
||||
}
|
||||
meta := map[string]string{}
|
||||
if err := json.Unmarshal([]byte(metaJSON), &meta); err != nil {
|
||||
return nil, fmt.Errorf("memory: unmarshal meta for %q: %w", id, err)
|
||||
}
|
||||
out = append(out, memory.Record{ID: id, Vec: decodeVec(blob), Meta: meta})
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("memory: rows: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Delete removes one vector by id. A row that is not there is not an error —
|
||||
// "forget this voice" is satisfied either way.
|
||||
func (m *MemoryStore) Delete(ctx context.Context, id string) error {
|
||||
if _, err := m.db.ExecContext(ctx, `DELETE FROM memory_vectors WHERE id = ?`, id); err != nil {
|
||||
return fmt.Errorf("memory: delete %q: %w", id, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// escapeLike neutralises the LIKE wildcards in a literal prefix.
|
||||
func escapeLike(s string) string {
|
||||
r := strings.NewReplacer(`\`, `\\`, `%`, `\%`, `_`, `\_`)
|
||||
return r.Replace(s)
|
||||
}
|
||||
|
||||
// encodeVec serializes a float32 slice as little-endian IEEE-754 bytes (4 bytes
|
||||
// per element) for the BLOB column.
|
||||
func encodeVec(v []float32) []byte {
|
||||
|
||||
Reference in New Issue
Block a user