Ship voice enrolment, and report recognition as blocked (#255)
Maven can now be told who someone is. She cannot yet tell who is speaking, and this commit is careful to say so rather than pretend otherwise. What works: profiles are enrolled from several deliberately recorded samples, listed, and deleted. They live in the existing memory_vectors table under a "speaker:" id prefix, so there is no migration; what that needed was a wider interface than memory.Store, hence memory.Catalog with ByPrefix and Delete. Delete is the load-bearing half — a voiceprint someone asked to be rid of has to actually go, and a search-only store cannot do that. InMemoryStore.Insert became an upsert by id to match what the persistent store already did. What does not work, and why it is not faked: there is no speaker-embedding model on this box. Sixteen ggufs in /mnt/hdd1/llms, all text; no ECAPA, no x-vector, no titanet, no wespeaker, no .onnx anywhere under /mnt/hdd1. So newSpeakerEmbedder returns nil, internal/speaker falls back to speaker.Disabled, Identify answers ErrDisabled, and the daemon logs which half is off at startup. The plan's "simple MFCC + GMM" floor is refused in the package comment: MFCC cosine distance detects channel and loudness as much as voice, and a biometric that is confidently wrong writes false claims about named people into his memory. A bad floor is worse than none here. Refused as well, and the reason is in enroll.go's doc comment: the plan asked for unknown speakers to be enrolled on first interaction with a TTS "кто это?". There is no request shape in the protocol that could express that. Taking a biometric of whoever walks past the microphone does it to guests who are not party to the exchange, and a synthesised question into a room is not consent from whoever answers. Authority: enrolment is AuthStepUp, because it is a deliberate sit-down act that writes a biometric of a named person and never something done by voice mid-conversation. Deletion is one rung lower at AuthWrite, deliberately inverting the usual pattern — getting rid of a biometric must never be the harder half. Listing is AuthRead and never returns the vectors themselves. Off unless configured: no speaker block means the three methods answer ErrUnknownMethod, so a default box has no wire path that takes a voiceprint. make build and make test pass. Vikunja #255 Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TrVSBKe3RFDF4fGYKWYQnX
This commit is contained in:
@@ -212,6 +212,11 @@ type Config struct {
|
||||
// default. See CaptureConfig.
|
||||
Capture *CaptureConfig `json:"capture,omitempty"`
|
||||
|
||||
// Speaker — voice identification (Vikunja #255). nil / absent ⇒ no
|
||||
// voiceprint is ever computed and nobody can be enrolled. Enabling it needs
|
||||
// a speaker-embedding model, which is not on this box. See SpeakerConfig.
|
||||
Speaker *SpeakerConfig `json:"speaker,omitempty"`
|
||||
|
||||
// MCP — Model Context Protocol servers Maven connects OUT to (Vikunja
|
||||
// #251). nil / absent / no enabled server ⇒ no connection is made and no
|
||||
// tool is discovered, like every other capability that reaches outside the
|
||||
@@ -619,6 +624,45 @@ func (c *CaptureConfig) MaxDuration() time.Duration {
|
||||
return time.Duration(c.MaxMinutes) * time.Minute
|
||||
}
|
||||
|
||||
// SpeakerConfig — voice identification (internal/speaker,
|
||||
// docs/plans/10-speaker-recognition.md).
|
||||
//
|
||||
// Absent, or enabled=false, ⇒ no voiceprint is computed for any turn, the
|
||||
// enrolment methods do not exist, and nobody can be enrolled. A voiceprint is
|
||||
// biometric data about a person, so this one is off until someone typed a model
|
||||
// path on purpose.
|
||||
//
|
||||
// It cannot currently be turned on: there is no speaker-embedding model on this
|
||||
// box. See the plan document for what to download.
|
||||
type SpeakerConfig struct {
|
||||
// Enabled — may she work out who is speaking. Default false.
|
||||
Enabled bool `json:"enabled,omitempty"`
|
||||
|
||||
// ModelPath — an ECAPA-TDNN (or equivalent) speaker-embedding ONNX model.
|
||||
// Required; without it the recognizer runs disabled and says so once.
|
||||
ModelPath string `json:"model_path,omitempty"`
|
||||
|
||||
// LibPath — onnxruntime shared library, as for the text embedder. Empty ⇒
|
||||
// the same default the embedder block uses.
|
||||
LibPath string `json:"lib_path,omitempty"`
|
||||
|
||||
// Threshold — cosine similarity a match must beat. 0 ⇒
|
||||
// speaker.DefaultThreshold (0.7). Lower it and she starts calling guests by
|
||||
// his name, which is the expensive direction of this error.
|
||||
Threshold float64 `json:"threshold,omitempty"`
|
||||
|
||||
// MinSeconds — least speech an identification will look at. 0 ⇒
|
||||
// speaker.DefaultMinSeconds (2s).
|
||||
MinSeconds float64 `json:"min_seconds,omitempty"`
|
||||
}
|
||||
|
||||
// Recognizes reports whether voice identification should be wired. Safe on a
|
||||
// nil receiver, and false without a model path — enabled with nothing to embed
|
||||
// with is a misconfiguration, not a capability.
|
||||
func (s *SpeakerConfig) Recognizes() bool {
|
||||
return s != nil && s.Enabled && strings.TrimSpace(s.ModelPath) != ""
|
||||
}
|
||||
|
||||
// WeatherConfig configures the weather provider for voice queries.
|
||||
type WeatherConfig struct {
|
||||
Provider string `json:"provider,omitempty"` // "open-meteo" or "" → stub
|
||||
|
||||
@@ -22,6 +22,9 @@ func TestSensesOffByDefault(t *testing.T) {
|
||||
if cfg.Capture.MaxDuration() != 0 {
|
||||
t.Error("a nil capture block invented a duration")
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("speaker recognition is on with no speaker block")
|
||||
}
|
||||
}
|
||||
|
||||
// The recorder is the capability that most needs its default to be off, so it
|
||||
@@ -158,3 +161,63 @@ func TestMediaWithoutVisionIsValid(t *testing.T) {
|
||||
t.Error("vision came on by itself")
|
||||
}
|
||||
}
|
||||
|
||||
// A voiceprint is a biometric of a named person. Nothing about it turns on by
|
||||
// itself: no speaker block means no recognition, and no enrolment either.
|
||||
func TestSpeakerIsOffUntilExplicitlyEnabled(t *testing.T) {
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(`{}`), &cfg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("speaker recognition came on with no config at all")
|
||||
}
|
||||
var empty Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{}}`), &empty); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if empty.Speaker.Recognizes() {
|
||||
t.Error("an empty speaker block enabled recognition")
|
||||
}
|
||||
}
|
||||
|
||||
// Enabled alone is not enough: recognition needs a model, and on this box there
|
||||
// is none. Recognizes() must stay false so the daemon reports the honest state
|
||||
// instead of claiming a capability it cannot perform.
|
||||
func TestSpeakerNeedsBothEnabledAndAModel(t *testing.T) {
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{"enabled":true}}`), &cfg); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if cfg.Speaker.Recognizes() {
|
||||
t.Error("enabled with no model_path claimed to recognise")
|
||||
}
|
||||
var only Config
|
||||
if err := json.Unmarshal([]byte(`{"speaker":{"model_path":"/opt/x.onnx"}}`), &only); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if only.Speaker.Recognizes() {
|
||||
t.Error("a model_path alone enabled recognition")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSpeakerBlockParsesFromJSON(t *testing.T) {
|
||||
const raw = `{"speaker":{"enabled":true,"model_path":"/opt/maven/models/spk/ecapa.onnx",` +
|
||||
`"lib_path":"/opt/maven/lib","threshold":0.62,"min_seconds":1.5}}`
|
||||
var cfg Config
|
||||
if err := json.Unmarshal([]byte(raw), &cfg); err != nil {
|
||||
t.Fatalf("unmarshal: %v", err)
|
||||
}
|
||||
if !cfg.Speaker.Recognizes() {
|
||||
t.Fatal("speaker did not parse as enabled")
|
||||
}
|
||||
if cfg.Speaker.ModelPath != "/opt/maven/models/spk/ecapa.onnx" {
|
||||
t.Errorf("model_path = %q", cfg.Speaker.ModelPath)
|
||||
}
|
||||
if cfg.Speaker.LibPath != "/opt/maven/lib" {
|
||||
t.Errorf("lib_path = %q", cfg.Speaker.LibPath)
|
||||
}
|
||||
if cfg.Speaker.Threshold != 0.62 || cfg.Speaker.MinSeconds != 1.5 {
|
||||
t.Errorf("thresholds = %+v", cfg.Speaker)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user