diff --git a/cmd/mavend/main.go b/cmd/mavend/main.go index 5fa9463..6e3db2b 100644 --- a/cmd/mavend/main.go +++ b/cmd/mavend/main.go @@ -169,6 +169,7 @@ func run(args []string) error { coreAPI ipc.CoreAPI eco *ecosystemWiring factWorker *factEnrichmentWorker + evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default) ) if !locked { @@ -263,6 +264,7 @@ func run(args []string) error { autotuneInterval := time.Duration(cfg.AutotuneInterval) tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals) factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval)) + evalWorker = newMemoryEvalWorker(st, phr, cfg) coreAPI = &daemonAPI{ CoreAPI: ipc.NewStoreAPI(st), @@ -440,6 +442,7 @@ func run(args []string) error { autotuneInterval := time.Duration(cfg.AutotuneInterval) tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals) factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval)) + evalWorker = newMemoryEvalWorker(st, phr, cfg) // Swap the CoreAPI from the locked placeholder to the real store adapter. newAPI := &daemonAPI{ @@ -476,6 +479,13 @@ func run(args []string) error { factWorker.run(ctx) }() + // Start background memory evaluation (nil unless configured). + if evalWorker != nil { + go func() { + evalWorker.run(ctx) + }() + } + dl.unlock() log.Printf("mavend: unlocked via passkey assertion") return nil @@ -514,6 +524,13 @@ func run(args []string) error { defer wg.Done() factWorker.run(ctx) }() + if evalWorker != nil { + wg.Add(1) + go func() { + defer wg.Done() + evalWorker.run(ctx) + }() + } } <-ctx.Done() diff --git a/cmd/mavend/memoryeval.go b/cmd/mavend/memoryeval.go new file mode 100644 index 0000000..9f0c641 --- /dev/null +++ b/cmd/mavend/memoryeval.go @@ -0,0 +1,88 @@ +// mavend/memoryeval.go — the driver for background memory evaluation +// (Vikunja #248). The evaluator itself is pure-ish and lives in +// internal/memeval; this is the one impure part: a ticker, the store, and the +// resident model's base URL. +// +// It is its own goroutine and NOT a step on the main tick, deliberately. The +// tick runs every 60s and has a delivery deadline behind it; an evaluation is +// a multi-second LLM round-trip on the same llama-server that answers voice +// turns, and it happens hourly at most. Bolting it onto the tick would make +// every hour's tick the slow one for no benefit. +package main + +import ( + "context" + "log" + "time" + + "github.com/kami/maven/internal/config" + "github.com/kami/maven/internal/llm" + "github.com/kami/maven/internal/memeval" + "github.com/kami/maven/internal/phraser" + "github.com/kami/maven/internal/store" +) + +// memoryEvalWorker — ticker + evaluator. +type memoryEvalWorker struct { + eval *memeval.Evaluator + interval time.Duration +} + +// newMemoryEvalWorker wires the evaluation loop, or returns nil when it should +// not run at all. nil is the normal case and every caller must handle it: +// +// - no memory_eval config block ⇒ off (a capability is off unless configured); +// - no LLM phraser ⇒ nothing to evaluate with. There is no template fallback +// here on purpose: a "memory evaluation" assembled from string templates +// would be a fixed sentence pretending to be an observation. +func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Config) *memoryEvalWorker { + if cfg.MemoryEval == nil { + return nil + } + lp, ok := phr.(*phraser.LLMPhraser) + if !ok { + log.Printf("memory eval: configured but no llama-server phraser — evaluation disabled") + return nil + } + interval := time.Duration(cfg.MemoryEval.Interval) + if interval <= 0 { + interval = config.DefaultMemoryEvalInterval + } + // A generous per-request timeout: this is a long prompt to a Thinking model + // and nobody is waiting on the answer. + client := llm.New(lp.BaseURL(), 5*time.Minute) + ev := memeval.NewEvaluator(st, st, client, memeval.Config{ + MaxItems: cfg.MemoryEval.MaxItems, + MinConfidence: cfg.MemoryEval.MinConfidence, + ContextBlock: contextBlockFn(cfg, time.Now), + }) + log.Printf("memory eval: enabled, every %s", interval) + return &memoryEvalWorker{eval: ev, interval: interval} +} + +// run evaluates every interval until ctx is canceled. +// +// The first evaluation waits a full interval rather than firing at startup, the +// opposite of the tick loop's cold-start behaviour. A tick that fires late is a +// nudge that arrives late; an evaluation that fires late is nothing at all, and +// the alternative is a heavy LLM call competing with startup — including with +// the first voice turn after a restart. +func (w *memoryEvalWorker) run(ctx context.Context) { + ticker := time.NewTicker(w.interval) + defer ticker.Stop() + for { + select { + case <-ctx.Done(): + return + case now := <-ticker.C: + obs, err := w.eval.Evaluate(ctx, now) + if err != nil { + log.Printf("memory eval: %v", err) + continue + } + for _, o := range obs { + log.Printf("memory eval: noted (%.2f, %s): %s", o.Conf, o.Action, o.Text) + } + } + } +} diff --git a/docs/plans/03-memory-evaluation.md b/docs/plans/03-memory-evaluation.md index a58be76..2dfedde 100644 --- a/docs/plans/03-memory-evaluation.md +++ b/docs/plans/03-memory-evaluation.md @@ -25,3 +25,46 @@ 6. Add `/eval` API method to `ipc.CoreAPI` (or reuse `Chat` with system context) so mavweb can show evaluation history 7. Add `memory_eval` block to `deploy/mavend.json` 8. Test with synthetic store state — verify observations match expected patterns + +--- + +## Status 2026-08-01 — foundation shipped (Vikunja #248) + +**Shipped:** `internal/memeval` (not `internal/memory/eval.go` — `internal/store` +imports `internal/memory` for the vector backend, so an evaluator that reads +`store.Fact` there would close an import cycle). `Evaluator.Evaluate` reads +`RecentFacts` / `RecentNotes` / `RecentNudges`, prompts the resident model under +a GBNF grammar for at most three `{observation, confidence, suggested_action}` +objects, drops anything under `min_confidence`, deduplicates against what earlier +evaluations wrote, and records the rest as notes with source `infer:memory-eval`. +Driver: `cmd/mavend/memoryeval.go`, its own goroutine on its own ticker. Config: +the `memory_eval` block — **absent ⇒ the loop does not run**. Visibility: `/dash` +already renders notes with their source, so evaluation output is visible with no +UI change. + +**Deliberately not shipped — this is policy, not an unfinished edge:** + +- *Dispatching observations as care nudges (plan step 4).* An hourly LLM loop + with permission to speak is a machine for generating interruptions, and the + content is model-generated text about his own life. The evaluator has no + dispatcher reference at all, so it cannot reach a channel by accident. Wiring + it to `delivery.Dispatcher` is a separate decision with its own opt-in. +- *Acting on `suggested_action`.* It is recorded inside the note text and + interpreted by nobody. No reminder, routine or fact is created. +- *Writing observation embeddings.* Notes are written with a nil embedding, so + they stay out of the RAG recall pool. Feeding generated text back into the pool + it came from is how a small model starts citing its own guesses as evidence. + +**Deferred, wants a decision or another capability:** + +- *Plan step 6, the `/eval` IPC method and an evaluation-history view.* `/dash` + covers reading the output; a dedicated trace surface is worth building once + there is real output to look at, and it should probably show the prompt too. +- *`RecentEvents`.* The plan lists it; the evaluator reads facts, notes and + nudges. Detected action/object events already drive pattern proposals (#43), and + duplicating them here would mostly re-derive that. +- *Output quality is unmeasured.* There is no fixture for "did she notice + something true". The tests cover the machinery — empty store, confidence floor, + dedupe, own-notes exclusion, error handling — not the observations. Until + someone reads a week of real output on `/dash`, treat the wording and the + `min_confidence` default as unvalidated. diff --git a/internal/config/config.go b/internal/config/config.go index b28b228..8883f61 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -146,6 +146,10 @@ type Config struct { // PatternProposalConfig. PatternProposals *PatternProposalConfig `json:"pattern_proposals,omitempty"` + // MemoryEval — background memory evaluation (internal/memeval). nil / + // absent ⇒ no evaluation loop at all. See MemoryEvalConfig. + MemoryEval *MemoryEvalConfig `json:"memory_eval,omitempty"` + // Praxis — the ecosystem attention-state service. When configured, maven // calls the Praxis HTTP tools API for attention listing and item lifecycle. // Maven never touches Praxis's database directly (ecosystem invariant: no @@ -392,6 +396,28 @@ func (p *PatternProposalConfig) AnnounceProposals() bool { return p != nil && p.Notify } +// MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248). +// Absent ⇒ off, like every other capability that costs something the owner did +// not ask for. Each evaluation is a full LLM round-trip on the one resident +// model, which is the same model answering him; running it hourly by default +// would put a multi-second stall in front of an occasional voice turn for a +// feature he may not want. +// +// The loop only ever writes notes (source infer:memory-eval, visible on +// /dash). It cannot speak — see internal/memeval. +type MemoryEvalConfig struct { + // Interval — how often to evaluate. 0 ⇒ DefaultMemoryEvalInterval. + Interval Duration `json:"interval,omitempty"` + + // MaxItems — recent facts / notes / nudges fed into one evaluation. + // 0 ⇒ memeval.DefaultMaxItems. + MaxItems int `json:"max_items,omitempty"` + + // MinConfidence — observations the model scores below this are dropped. + // 0 ⇒ memeval.DefaultMinConfidence. + MinConfidence float64 `json:"min_confidence,omitempty"` +} + // PhraserConfig — the LLM-backed phraser seam. The daemon spawns llama-server // as a managed subprocess and sends chat-completion requests to phrase nudge // and reminder messages. nil ⇒ the template-based Stub is used instead. @@ -493,6 +519,10 @@ const ( // most. A proposal is never urgent; if two patterns surface in the same // hour, the second one waits, and the /routines page has it either way. DefaultProposalCooldown = 24 * time.Hour + + // DefaultMemoryEvalInterval — the plan's cadence (1h) for the memory + // evaluation loop, applied only when the block is present at all. + DefaultMemoryEvalInterval = time.Hour ) // Load reads the JSON config at path and applies defaults. A missing file is @@ -571,6 +601,12 @@ func (c *Config) applyDefaults() { c.PatternProposals.Cooldown = Duration(DefaultProposalCooldown) } + // Same rule: absent stays nil (⇒ no evaluation loop), present gets defaults + // so `{}` is a valid "on with the plan's cadence". + if c.MemoryEval != nil && c.MemoryEval.Interval <= 0 { + c.MemoryEval.Interval = Duration(DefaultMemoryEvalInterval) + } + if c.Voice != nil { if c.Voice.RouterThreshold <= 0 { c.Voice.RouterThreshold = DefaultRouterThreshold diff --git a/internal/config/config_test.go b/internal/config/config_test.go index dae1b6b..9f25ed1 100644 --- a/internal/config/config_test.go +++ b/internal/config/config_test.go @@ -243,3 +243,53 @@ func TestDurationRoundTrip(t *testing.T) { t.Errorf("round-trip = %v, want %v", d2, d) } } + +// Both new opt-in capabilities follow the same rule: absent block ⇒ nil ⇒ the +// behaviour does not exist. Presence is the enable act, so a bare `{}` block is +// valid and gets the defaults filled in. +func TestOptInBlocksAbsentStayNil(t *testing.T) { + c, err := Load(writeConfig(t, `{}`)) + if err != nil { + t.Fatalf("Load: %v", err) + } + if c.PatternProposals != nil { + t.Errorf("pattern_proposals absent but got %+v", c.PatternProposals) + } + if c.PatternProposals.AnnounceProposals() { + t.Error("AnnounceProposals() true with no config block") + } + if c.MemoryEval != nil { + t.Errorf("memory_eval absent but got %+v", c.MemoryEval) + } +} + +func TestOptInBlocksGetDefaultsWhenPresent(t *testing.T) { + c, err := Load(writeConfig(t, `{"pattern_proposals":{"notify":true},"memory_eval":{}}`)) + if err != nil { + t.Fatalf("Load: %v", err) + } + if !c.PatternProposals.AnnounceProposals() { + t.Error("notify:true did not enable announcements") + } + if time.Duration(c.PatternProposals.Cooldown) != DefaultProposalCooldown { + t.Errorf("proposal cooldown = %v, want %v", c.PatternProposals.Cooldown, DefaultProposalCooldown) + } + if time.Duration(c.MemoryEval.Interval) != DefaultMemoryEvalInterval { + t.Errorf("memory eval interval = %v, want %v", c.MemoryEval.Interval, DefaultMemoryEvalInterval) + } +} + +// Notify is off even when the block exists — the block is where you tune it, +// notify:true is the act that lets her speak. +func TestPatternProposalNotifyDefaultsOff(t *testing.T) { + c, err := Load(writeConfig(t, `{"pattern_proposals":{"cooldown":"6h"}}`)) + if err != nil { + t.Fatalf("Load: %v", err) + } + if c.PatternProposals.AnnounceProposals() { + t.Error("notify defaulted to on") + } + if time.Duration(c.PatternProposals.Cooldown) != 6*time.Hour { + t.Errorf("cooldown = %v, want 6h", c.PatternProposals.Cooldown) + } +} diff --git a/internal/memeval/eval.go b/internal/memeval/eval.go new file mode 100644 index 0000000..1edb5f5 --- /dev/null +++ b/internal/memeval/eval.go @@ -0,0 +1,357 @@ +// Package memeval is background memory evaluation (Vikunja #248, +// docs/plans/03-memory-evaluation.md). +// +// It lives beside internal/memory rather than inside it because +// internal/store imports internal/memory for the vector-store backend, and an +// evaluator has to read store.Fact / store.Note / store.Nudge — putting it in +// internal/memory would close that import cycle. +// +// Every so often Maven reads back her own recent memory — facts, notes, the +// nudges she sent — and asks the resident model what it notices: a habit that +// stopped, a gap, something worth saying later. What comes back is written as +// notes with source EvalNoteSource and nothing else happens. That restraint is +// the design, not an unfinished edge: +// +// - She does not speak here. There is no dispatcher, no channel, no nudge. +// An observation is a thought she wrote down; he reads it on /dash when he +// wants to. "Not a nag, not autonomous" (CLAUDE.md) is easy to violate with +// exactly this feature — an hourly loop with an LLM in it and permission to +// talk is a machine for generating interruptions — so the loop has no way +// to reach him at all. Turning observations into nudges is a separate +// decision with a separate opt-in, and it is deliberately NOT in this file. +// - She does not act. No reminder is created, no routine proposed, no fact +// written. The model's suggested_action is recorded as text inside the note +// and interpreted by nobody. +// - She says nothing about an empty store. No memory ⇒ no LLM call ⇒ no +// "observations" invented out of two facts. A 1.7B asked to find a pattern +// will always find one; the defence is not asking. +// +// Everything the evaluator writes is attributable: source is EvalNoteSource, so +// an inferred observation can never be mistaken for something he said, and the +// whole batch is one SQL delete away if the output turns out to be noise. +package memeval + +import ( + "context" + "encoding/json" + "fmt" + "sort" + "strings" + "time" + + "github.com/kami/maven/internal/llm" + "github.com/kami/maven/internal/persona" + "github.com/kami/maven/internal/store" +) + +// EvalNoteSource — the source stamped on every note the evaluator writes. +// Same infer:* convention as the rest of the derived facts. +const EvalNoteSource = "infer:memory-eval" + +// DefaultMinConfidence — an observation below this is dropped. The model is +// asked for its own confidence and small models are badly calibrated, so this +// is a coarse filter, not a probability: it exists to throw away the guesses +// the model itself hedged on. +const DefaultMinConfidence = 0.7 + +// DefaultMaxItems — how much recent memory goes into one evaluation, per +// store. 30 facts + 30 notes + 30 nudges is a few thousand tokens of the 4096 +// context the resident Thinking model runs with, which leaves room for its +// reasoning tokens. Raising this trades reasoning room for history. +const DefaultMaxItems = 30 + +// MaxObservations — the model may return at most this many observations per +// evaluation, enforced by the grammar. A cap here is also a noise cap: an +// evaluation that "notices" ten things has noticed nothing. +const MaxObservations = 3 + +// Observation — one thing the evaluator noticed. +type Observation struct { + Text string `json:"observation"` + Conf float64 `json:"confidence"` + // Action — what the model thinks should happen with this. Recorded, never + // executed: see the file comment. One of "note", "propose", "notify". + Action string `json:"suggested_action"` +} + +// Completer — the llama-server seam, same shape router.Completer uses so the +// one resident model serves this caller too. +type Completer interface { + Complete(ctx context.Context, r llm.Req) (string, error) +} + +// Reader — the slice of the store an evaluation reads. Narrow on purpose: the +// evaluator gets recent memory and nothing else. No entity graph, no presence, +// no config facts. +type Reader interface { + RecentFacts(ctx context.Context, n int) ([]store.Fact, error) + RecentNotes(ctx context.Context, n int) ([]store.Note, error) + RecentNudges(ctx context.Context, n int) ([]store.Nudge, error) +} + +// NoteWriter — where observations land. Embeddings are passed nil: an +// observation is written for a human to read on /dash, not to be recalled by +// similarity. Feeding LLM-generated text back into the RAG pool it was +// generated from is how a small model starts citing its own guesses as +// evidence. +type NoteWriter interface { + WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) +} + +// Config — evaluator tuning. Zero values are replaced by the Default* +// constants, so the zero Config is the sane one. +type Config struct { + MaxItems int + MinConfidence float64 + + // ContextBlock — the shared persona block (internal/persona), re-evaluated + // per call so the clock in it is current. Prepended to the system prompt so + // observations come out in Maven's voice: feminine self-reference, informal + // "ты". nil is allowed; the base prompt still carries the address rules. + ContextBlock func() string +} + +// Evaluator reads recent memory and records what the model notices. +type Evaluator struct { + read Reader + write NoteWriter + llm Completer + cfg Config +} + +func NewEvaluator(r Reader, w NoteWriter, c Completer, cfg Config) *Evaluator { + if cfg.MaxItems <= 0 { + cfg.MaxItems = DefaultMaxItems + } + if cfg.MinConfidence <= 0 { + cfg.MinConfidence = DefaultMinConfidence + } + return &Evaluator{read: r, write: w, llm: c, cfg: cfg} +} + +// evalGrammar — GBNF pinning the reply to a bounded JSON array of fixed-shape +// observations. Same reasoning as the router's routeGrammar: the enum and the +// length bound are what stop a small model from drifting into free text or +// filling the token budget with one repeated field. +const evalGrammar = ` +root ::= "[" ws (obs ("," ws obs){0,2})? ws "]" +obs ::= "{" ws "\"observation\"" ws ":" ws text "," ws "\"confidence\"" ws ":" ws conf "," ws "\"suggested_action\"" ws ":" ws act ws "}" +text ::= "\"" ([^"\\] | "\\" .){1,200} "\"" +conf ::= "0" "." [0-9]{1,2} | "1" ("." "0")? +act ::= "\"note\"" | "\"propose\"" | "\"notify\"" +ws ::= [ \t\n]* +` + +// evalSystem — the evaluation prompt. Two things it insists on, both learned +// from the phraser: state the observation as something she noticed rather than +// an instruction, and say nothing when there is nothing (the model is given an +// explicit way to return an empty array, because a model with no exit returns +// filler). +const evalSystem = `Ты просматриваешь свою собственную память: недавние факты, заметки и напоминания, которые ты отправляла. +Найди то, что действительно заметно: привычка, которая прервалась; пробел в записях; повторяющаяся закономерность. + +Правила: +- Отвечай ТОЛЬКО массивом JSON. Каждый элемент: {"observation": "...", "confidence": 0.0-1.0, "suggested_action": "note"|"propose"|"notify"}. +- observation — короткая фраза по-русски о том, что ты заметила. О себе — в женском роде ("я заметила"). К нему — на "ты". +- Не выдумывай. Если в памяти нет ничего заметного, верни пустой массив []. +- Не давай советов и не приказывай. Ты замечаешь, а не требуешь. +- confidence — насколько ты уверена, что это настоящая закономерность, а не совпадение. +- Максимум три наблюдения. Лучше одно точное, чем три общих.` + +// Evaluate runs one evaluation and returns the observations it recorded. +// +// Returns (nil, nil) — not an error — for every ordinary "nothing to say" +// outcome: an empty store, an empty array from the model, everything below the +// confidence floor, or every observation already recorded earlier. Only a real +// read/LLM/write failure is an error, and the caller (a background ticker) logs +// it and waits for the next interval. +func (e *Evaluator) Evaluate(ctx context.Context, now time.Time) ([]Observation, error) { + snap, err := e.snapshot(ctx) + if err != nil { + return nil, err + } + if snap == "" { + return nil, nil // nothing recorded ⇒ nothing to notice, and no LLM call + } + + raw, err := e.llm.Complete(ctx, llm.Req{ + System: persona.Prepend(e.cfg.ContextBlock, evalSystem), + User: snap, + Grammar: evalGrammar, + MaxTokens: 512, + RepeatPenalty: 1.1, + }) + if err != nil { + return nil, fmt.Errorf("memory eval: complete: %w", err) + } + obs, err := parseObservations(raw) + if err != nil { + return nil, fmt.Errorf("memory eval: parse %q: %w", truncate(raw, 120), err) + } + + // Dedupe against what earlier evaluations already wrote. Without this an + // hourly loop over a slowly-changing store writes the same sentence every + // hour until /dash is nothing but the evaluator talking to itself. + seen, err := e.recordedTexts(ctx) + if err != nil { + return nil, err + } + + var kept []Observation + for _, o := range obs { + o.Text = strings.TrimSpace(o.Text) + if o.Text == "" || o.Conf < e.cfg.MinConfidence { + continue + } + norm := normalizeObservation(o.Text) + if seen[norm] { + continue + } + seen[norm] = true + if _, err := e.write.WriteNote(ctx, now, formatNote(o), nil, EvalNoteSource); err != nil { + return kept, fmt.Errorf("memory eval: write note: %w", err) + } + kept = append(kept, o) + } + return kept, nil +} + +// formatNote — the stored text. The suggested action is kept as a visible +// suffix rather than a column: it is the model's opinion about what to do next, +// and the only consumer is a human reading /dash. +func formatNote(o Observation) string { + if o.Action == "" { + return o.Text + } + return fmt.Sprintf("%s [%s]", o.Text, o.Action) +} + +// recordedTexts — the normalized text of every observation earlier evaluations +// wrote, for dedupe. Reads a wider window than MaxItems because the point is to +// remember saying it, not to summarize it. +func (e *Evaluator) recordedTexts(ctx context.Context) (map[string]bool, error) { + notes, err := e.read.RecentNotes(ctx, 200) + if err != nil { + return nil, fmt.Errorf("memory eval: recent notes: %w", err) + } + seen := make(map[string]bool, len(notes)) + for _, n := range notes { + if n.Source != EvalNoteSource { + continue + } + text := n.Text + // Strip the "[action]" suffix formatNote appended. + if i := strings.LastIndex(text, " ["); i > 0 && strings.HasSuffix(text, "]") { + text = text[:i] + } + seen[normalizeObservation(text)] = true + } + return seen, nil +} + +// normalizeObservation — dedupe key. Case- and whitespace-insensitive, which +// catches the realistic repeat (the model re-emitting the same sentence with a +// different comma) without pretending to do semantic dedupe. +func normalizeObservation(s string) string { + return strings.Join(strings.Fields(strings.ToLower(s)), " ") +} + +// snapshot renders recent memory as the user turn. Returns "" when there is +// nothing in any store — the caller treats that as "do not ask the model". +// +// Notes written by earlier evaluations are excluded. Feeding her own +// observations back in is how "я заметила, что ты не записывал еду" becomes +// evidence for noticing it again, three evaluations deep. +func (e *Evaluator) snapshot(ctx context.Context) (string, error) { + n := e.cfg.MaxItems + facts, err := e.read.RecentFacts(ctx, n) + if err != nil { + return "", fmt.Errorf("memory eval: recent facts: %w", err) + } + notes, err := e.read.RecentNotes(ctx, n) + if err != nil { + return "", fmt.Errorf("memory eval: recent notes: %w", err) + } + nudges, err := e.read.RecentNudges(ctx, n) + if err != nil { + return "", fmt.Errorf("memory eval: recent nudges: %w", err) + } + + var b strings.Builder + wrote := false + if len(facts) > 0 { + b.WriteString("Факты:\n") + for _, f := range facts { + fmt.Fprintf(&b, "- %s %s=%s (%s)\n", f.Ts.Format("2006-01-02 15:04"), f.Key, truncate(f.Value, 80), f.Source) + wrote = true + } + } + own := 0 + var noteLines []string + for _, nt := range notes { + if nt.Source == EvalNoteSource { + own++ + continue + } + noteLines = append(noteLines, fmt.Sprintf("- %s %s\n", nt.Ts.Format("2006-01-02 15:04"), truncate(nt.Text, 160))) + } + if len(noteLines) > 0 { + b.WriteString("\nЗаметки:\n") + for _, l := range noteLines { + b.WriteString(l) + wrote = true + } + } + if len(nudges) > 0 { + b.WriteString("\nНапоминания, которые ты отправляла:\n") + for _, nd := range nudges { + outcome := nd.Outcome + if outcome == "" { + outcome = "?" + } + fmt.Fprintf(&b, "- %s %s → %s (%s)\n", nd.Ts.Format("2006-01-02 15:04"), nd.Rule, outcome, nd.Channel) + wrote = true + } + } + if !wrote { + // Only her own past observations, or nothing at all. Either way there is + // no new memory to evaluate. + return "", nil + } + b.WriteString("\nЧто ты замечаешь?") + return b.String(), nil +} + +// parseObservations reads the model's array. Tolerates the leading/trailing +// prose a Thinking model sometimes emits around JSON by taking the outermost +// bracketed span, the same tolerance the router's parser has. +func parseObservations(raw string) ([]Observation, error) { + s := strings.TrimSpace(raw) + if i := strings.Index(s, "["); i >= 0 { + if j := strings.LastIndex(s, "]"); j > i { + s = s[i : j+1] + } + } + if s == "" { + return nil, nil + } + var obs []Observation + if err := json.Unmarshal([]byte(s), &obs); err != nil { + return nil, err + } + if len(obs) > MaxObservations { + // The grammar bounds this; a grammar-less server or a future prompt + // change must not be able to flood /dash. + sort.SliceStable(obs, func(i, j int) bool { return obs[i].Conf > obs[j].Conf }) + obs = obs[:MaxObservations] + } + return obs, nil +} + +func truncate(s string, n int) string { + r := []rune(s) + if len(r) <= n { + return s + } + return string(r[:n]) + "…" +} diff --git a/internal/memeval/eval_test.go b/internal/memeval/eval_test.go new file mode 100644 index 0000000..93979ee --- /dev/null +++ b/internal/memeval/eval_test.go @@ -0,0 +1,271 @@ +package memeval + +import ( + "context" + "database/sql" + "errors" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/kami/maven/internal/llm" + "github.com/kami/maven/internal/store" +) + +// fakeLLM — canned replies, one per call, and a record of what it was asked. +type fakeLLM struct { + replies []string + calls []llm.Req + err error +} + +func (f *fakeLLM) Complete(_ context.Context, r llm.Req) (string, error) { + f.calls = append(f.calls, r) + if f.err != nil { + return "", f.err + } + if len(f.replies) == 0 { + return "[]", nil + } + out := f.replies[0] + f.replies = f.replies[1:] + return out, nil +} + +func newTestStore(t *testing.T) *store.Store { + t.Helper() + st, err := store.Open(context.Background(), filepath.Join(t.TempDir(), "memeval_test.db")) + if err != nil { + t.Fatalf("store.Open: %v", err) + } + t.Cleanup(func() { _ = st.Close() }) + return st +} + +func refNow() time.Time { return time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC) } + +// seedMemory writes a little of everything the evaluator reads. +func seedMemory(t *testing.T, st *store.Store, ctx context.Context, now time.Time) { + t.Helper() + for i := 0; i < 3; i++ { + ts := now.Add(-time.Duration(i+1) * 24 * time.Hour) + if _, err := st.WriteFact(ctx, ts, store.KindSelf, "water_ml", "500", "tap:desk", 1.0, sql.NullInt64{}); err != nil { + t.Fatalf("write fact: %v", err) + } + } + if _, err := st.WriteNote(ctx, now.Add(-2*time.Hour), "купить корм для кота", nil, "tap:voice"); err != nil { + t.Fatalf("write note: %v", err) + } + if _, err := st.RecordNudge(ctx, "water", "voice", "пора выпить воды", now.Add(-time.Hour)); err != nil { + t.Fatalf("record nudge: %v", err) + } +} + +// TestEvaluateEmptyStoreAsksNothing — the "shuts up when uncertain" floor. An +// empty store must not even reach the model: a small model asked to find a +// pattern in nothing will invent one. +func TestEvaluateEmptyStoreAsksNothing(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + f := &fakeLLM{} + ev := NewEvaluator(st, st, f, Config{}) + + obs, err := ev.Evaluate(ctx, refNow()) + if err != nil { + t.Fatalf("Evaluate: %v", err) + } + if len(obs) != 0 { + t.Fatalf("observations on an empty store = %d, want 0", len(obs)) + } + if len(f.calls) != 0 { + t.Fatalf("LLM called %d times on an empty store, want 0", len(f.calls)) + } +} + +// TestEvaluateWritesHighConfidenceObservations — the happy path. Confident +// observations are written as notes stamped infer:memory-eval, and the low +// ones are dropped. +func TestEvaluateWritesHighConfidenceObservations(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + now := refNow() + seedMemory(t, st, ctx, now) + + f := &fakeLLM{replies: []string{`[ + {"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"notify"}, + {"observation":"может быть, ты стал меньше пить воды","confidence":0.3,"suggested_action":"note"} + ]`}} + ev := NewEvaluator(st, st, f, Config{}) + + obs, err := ev.Evaluate(ctx, now) + if err != nil { + t.Fatalf("Evaluate: %v", err) + } + if len(obs) != 1 { + t.Fatalf("kept %d observations, want 1 (the 0.3 one is below the floor): %+v", len(obs), obs) + } + if obs[0].Text != "ты три дня не записывал еду" { + t.Errorf("kept the wrong observation: %q", obs[0].Text) + } + + notes, err := st.RecentNotes(ctx, 50) + if err != nil { + t.Fatalf("RecentNotes: %v", err) + } + var written []store.Note + for _, n := range notes { + if n.Source == EvalNoteSource { + written = append(written, n) + } + } + if len(written) != 1 { + t.Fatalf("notes with source %s = %d, want 1", EvalNoteSource, len(written)) + } + if !strings.Contains(written[0].Text, "ты три дня не записывал еду") { + t.Errorf("note text = %q", written[0].Text) + } + if !strings.Contains(written[0].Text, "[notify]") { + t.Errorf("note text = %q, want the suggested action recorded", written[0].Text) + } + + // The prompt must carry the memory it is evaluating, and must not carry a + // grammar-free request. + if len(f.calls) != 1 { + t.Fatalf("LLM calls = %d, want 1", len(f.calls)) + } + if !strings.Contains(f.calls[0].User, "water_ml") { + t.Errorf("prompt does not mention the seeded facts:\n%s", f.calls[0].User) + } + if f.calls[0].Grammar == "" { + t.Error("evaluation ran without a grammar") + } +} + +// TestEvaluateDeduplicatesAcrossRuns — the failure mode that would make this +// feature unusable: an hourly loop over a store that barely changes writing the +// same sentence every hour until /dash is nothing but the evaluator. +func TestEvaluateDeduplicatesAcrossRuns(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + now := refNow() + seedMemory(t, st, ctx, now) + + same := `[{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"note"}]` + spaced := `[{"observation":"Ты три дня не записывал еду","confidence":0.95,"suggested_action":"note"}]` + f := &fakeLLM{replies: []string{same, same, spaced}} + ev := NewEvaluator(st, st, f, Config{}) + + for i := 0; i < 3; i++ { + if _, err := ev.Evaluate(ctx, now.Add(time.Duration(i)*time.Hour)); err != nil { + t.Fatalf("Evaluate %d: %v", i, err) + } + } + + notes, err := st.RecentNotes(ctx, 50) + if err != nil { + t.Fatalf("RecentNotes: %v", err) + } + n := 0 + for _, nt := range notes { + if nt.Source == EvalNoteSource { + n++ + } + } + if n != 1 { + t.Fatalf("eval notes after three identical evaluations = %d, want 1", n) + } +} + +// TestEvaluateIgnoresOwnNotes — her own observations must not become input. +// Otherwise "я заметила X" is evidence for noticing X again, three evaluations +// deep. With nothing but eval notes in the store there is no new memory, so the +// model is not asked at all. +func TestEvaluateIgnoresOwnNotes(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + now := refNow() + if _, err := st.WriteNote(ctx, now.Add(-time.Hour), "я заметила, что ты мало пьёшь [note]", nil, EvalNoteSource); err != nil { + t.Fatalf("write note: %v", err) + } + + f := &fakeLLM{} + ev := NewEvaluator(st, st, f, Config{}) + obs, err := ev.Evaluate(ctx, now) + if err != nil { + t.Fatalf("Evaluate: %v", err) + } + if len(obs) != 0 || len(f.calls) != 0 { + t.Fatalf("observations=%d llm calls=%d, want 0/0 — own notes are not memory to evaluate", len(obs), len(f.calls)) + } +} + +// TestEvaluateEmptyArrayIsNotAnError — "nothing to say" is the expected outcome +// most of the time and must not be logged as a failure. +func TestEvaluateEmptyArrayIsNotAnError(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + now := refNow() + seedMemory(t, st, ctx, now) + + ev := NewEvaluator(st, st, &fakeLLM{replies: []string{"[]"}}, Config{}) + obs, err := ev.Evaluate(ctx, now) + if err != nil { + t.Fatalf("Evaluate: %v", err) + } + if len(obs) != 0 { + t.Fatalf("observations = %d, want 0", len(obs)) + } +} + +// TestEvaluateLLMErrorIsReported — a broken llama-server is an error the caller +// logs; it must not silently write anything. +func TestEvaluateLLMErrorIsReported(t *testing.T) { + st := newTestStore(t) + ctx := context.Background() + now := refNow() + seedMemory(t, st, ctx, now) + + ev := NewEvaluator(st, st, &fakeLLM{err: errors.New("connection refused")}, Config{}) + if _, err := ev.Evaluate(ctx, now); err == nil { + t.Fatal("want an error when the model is unreachable") + } + notes, err := st.RecentNotes(ctx, 50) + if err != nil { + t.Fatalf("RecentNotes: %v", err) + } + for _, n := range notes { + if n.Source == EvalNoteSource { + t.Fatalf("wrote a note despite an LLM failure: %q", n.Text) + } + } +} + +// TestParseObservationsTolerantAndBounded — Thinking models wrap JSON in prose, +// and no reply may exceed MaxObservations even if the grammar is bypassed. +func TestParseObservationsTolerantAndBounded(t *testing.T) { + obs, err := parseObservations(`hmm вот: [{"observation":"a","confidence":0.9,"suggested_action":"note"}] всё`) + if err != nil { + t.Fatalf("parse: %v", err) + } + if len(obs) != 1 || obs[0].Text != "a" { + t.Fatalf("got %+v, want one observation 'a'", obs) + } + + var b strings.Builder + b.WriteString("[") + for i := 0; i < MaxObservations+3; i++ { + if i > 0 { + b.WriteString(",") + } + b.WriteString(`{"observation":"x","confidence":0.5,"suggested_action":"note"}`) + } + b.WriteString("]") + obs, err = parseObservations(b.String()) + if err != nil { + t.Fatalf("parse: %v", err) + } + if len(obs) != MaxObservations { + t.Fatalf("parsed %d observations, want the %d cap", len(obs), MaxObservations) + } +}