Files
Maven/internal/memeval/eval_test.go
kami dc7c72a3d7 Add background memory evaluation, off unless configured (#248)
Ships the real, local, testable part of the memory-evaluation plan
(docs/plans/03-memory-evaluation.md): Maven reads back her own recent
memory on a slow ticker, asks the resident model what it notices, and
records the confident answers as notes.

internal/memeval — not internal/memory/eval.go as the plan says, because
internal/store imports internal/memory for the vector backend and an
evaluator has to read store.Fact/Note/Nudge, which would close the
cycle. Evaluate() gathers RecentFacts/RecentNotes/RecentNudges, prompts
under a GBNF grammar bounded to three {observation, confidence,
suggested_action} objects, drops anything under min_confidence,
deduplicates against what earlier runs wrote, and writes the rest as
notes with source infer:memory-eval. /dash already renders notes with
their source, so the output is visible with no UI change.

cmd/mavend/memoryeval.go drives it on its own goroutine and ticker, not
on the 60s tick: an evaluation is a multi-second round-trip on the same
llama-server that answers voice turns, and it runs hourly at most. The
memory_eval config block is absent by default and absence means the
goroutine does not exist. No llama-server phraser also means no loop —
there is no template fallback, because a "memory evaluation" assembled
from templates is a fixed sentence pretending to be an observation.

What it deliberately cannot do, since this is the feature most likely to
turn Maven into a nag:

  - It cannot speak. No dispatcher reference, no channel, no nudge. An
    observation is a thought she wrote down and he reads on /dash.
    Announcing them is a separate decision with its own opt-in.
  - It cannot act. suggested_action is recorded as text and interpreted
    by nobody — no reminder, routine or fact is created from it.
  - It says nothing about an empty store: no memory means no LLM call,
    so there are no observations invented out of two facts.
  - Its own notes are excluded from the next evaluation's input, and are
    written with a nil embedding so they stay out of the recall pool.

The plan's remaining items (dispatching observations, an /eval IPC
method and trace view, RecentEvents) and the fact that output quality is
entirely unmeasured are written up at the bottom of the plan doc.
2026-08-01 01:45:49 +04:00

272 lines
8.4 KiB
Go

package memeval
import (
"context"
"database/sql"
"errors"
"path/filepath"
"strings"
"testing"
"time"
"github.com/kami/maven/internal/llm"
"github.com/kami/maven/internal/store"
)
// fakeLLM — canned replies, one per call, and a record of what it was asked.
type fakeLLM struct {
replies []string
calls []llm.Req
err error
}
func (f *fakeLLM) Complete(_ context.Context, r llm.Req) (string, error) {
f.calls = append(f.calls, r)
if f.err != nil {
return "", f.err
}
if len(f.replies) == 0 {
return "[]", nil
}
out := f.replies[0]
f.replies = f.replies[1:]
return out, nil
}
func newTestStore(t *testing.T) *store.Store {
t.Helper()
st, err := store.Open(context.Background(), filepath.Join(t.TempDir(), "memeval_test.db"))
if err != nil {
t.Fatalf("store.Open: %v", err)
}
t.Cleanup(func() { _ = st.Close() })
return st
}
func refNow() time.Time { return time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC) }
// seedMemory writes a little of everything the evaluator reads.
func seedMemory(t *testing.T, st *store.Store, ctx context.Context, now time.Time) {
t.Helper()
for i := 0; i < 3; i++ {
ts := now.Add(-time.Duration(i+1) * 24 * time.Hour)
if _, err := st.WriteFact(ctx, ts, store.KindSelf, "water_ml", "500", "tap:desk", 1.0, sql.NullInt64{}); err != nil {
t.Fatalf("write fact: %v", err)
}
}
if _, err := st.WriteNote(ctx, now.Add(-2*time.Hour), "купить корм для кота", nil, "tap:voice"); err != nil {
t.Fatalf("write note: %v", err)
}
if _, err := st.RecordNudge(ctx, "water", "voice", "пора выпить воды", now.Add(-time.Hour)); err != nil {
t.Fatalf("record nudge: %v", err)
}
}
// TestEvaluateEmptyStoreAsksNothing — the "shuts up when uncertain" floor. An
// empty store must not even reach the model: a small model asked to find a
// pattern in nothing will invent one.
func TestEvaluateEmptyStoreAsksNothing(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
f := &fakeLLM{}
ev := NewEvaluator(st, st, f, Config{})
obs, err := ev.Evaluate(ctx, refNow())
if err != nil {
t.Fatalf("Evaluate: %v", err)
}
if len(obs) != 0 {
t.Fatalf("observations on an empty store = %d, want 0", len(obs))
}
if len(f.calls) != 0 {
t.Fatalf("LLM called %d times on an empty store, want 0", len(f.calls))
}
}
// TestEvaluateWritesHighConfidenceObservations — the happy path. Confident
// observations are written as notes stamped infer:memory-eval, and the low
// ones are dropped.
func TestEvaluateWritesHighConfidenceObservations(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
now := refNow()
seedMemory(t, st, ctx, now)
f := &fakeLLM{replies: []string{`[
{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"notify"},
{"observation":"может быть, ты стал меньше пить воды","confidence":0.3,"suggested_action":"note"}
]`}}
ev := NewEvaluator(st, st, f, Config{})
obs, err := ev.Evaluate(ctx, now)
if err != nil {
t.Fatalf("Evaluate: %v", err)
}
if len(obs) != 1 {
t.Fatalf("kept %d observations, want 1 (the 0.3 one is below the floor): %+v", len(obs), obs)
}
if obs[0].Text != "ты три дня не записывал еду" {
t.Errorf("kept the wrong observation: %q", obs[0].Text)
}
notes, err := st.RecentNotes(ctx, 50)
if err != nil {
t.Fatalf("RecentNotes: %v", err)
}
var written []store.Note
for _, n := range notes {
if n.Source == EvalNoteSource {
written = append(written, n)
}
}
if len(written) != 1 {
t.Fatalf("notes with source %s = %d, want 1", EvalNoteSource, len(written))
}
if !strings.Contains(written[0].Text, "ты три дня не записывал еду") {
t.Errorf("note text = %q", written[0].Text)
}
if !strings.Contains(written[0].Text, "[notify]") {
t.Errorf("note text = %q, want the suggested action recorded", written[0].Text)
}
// The prompt must carry the memory it is evaluating, and must not carry a
// grammar-free request.
if len(f.calls) != 1 {
t.Fatalf("LLM calls = %d, want 1", len(f.calls))
}
if !strings.Contains(f.calls[0].User, "water_ml") {
t.Errorf("prompt does not mention the seeded facts:\n%s", f.calls[0].User)
}
if f.calls[0].Grammar == "" {
t.Error("evaluation ran without a grammar")
}
}
// TestEvaluateDeduplicatesAcrossRuns — the failure mode that would make this
// feature unusable: an hourly loop over a store that barely changes writing the
// same sentence every hour until /dash is nothing but the evaluator.
func TestEvaluateDeduplicatesAcrossRuns(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
now := refNow()
seedMemory(t, st, ctx, now)
same := `[{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"note"}]`
spaced := `[{"observation":"Ты три дня не записывал еду","confidence":0.95,"suggested_action":"note"}]`
f := &fakeLLM{replies: []string{same, same, spaced}}
ev := NewEvaluator(st, st, f, Config{})
for i := 0; i < 3; i++ {
if _, err := ev.Evaluate(ctx, now.Add(time.Duration(i)*time.Hour)); err != nil {
t.Fatalf("Evaluate %d: %v", i, err)
}
}
notes, err := st.RecentNotes(ctx, 50)
if err != nil {
t.Fatalf("RecentNotes: %v", err)
}
n := 0
for _, nt := range notes {
if nt.Source == EvalNoteSource {
n++
}
}
if n != 1 {
t.Fatalf("eval notes after three identical evaluations = %d, want 1", n)
}
}
// TestEvaluateIgnoresOwnNotes — her own observations must not become input.
// Otherwise "я заметила X" is evidence for noticing X again, three evaluations
// deep. With nothing but eval notes in the store there is no new memory, so the
// model is not asked at all.
func TestEvaluateIgnoresOwnNotes(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
now := refNow()
if _, err := st.WriteNote(ctx, now.Add(-time.Hour), "я заметила, что ты мало пьёшь [note]", nil, EvalNoteSource); err != nil {
t.Fatalf("write note: %v", err)
}
f := &fakeLLM{}
ev := NewEvaluator(st, st, f, Config{})
obs, err := ev.Evaluate(ctx, now)
if err != nil {
t.Fatalf("Evaluate: %v", err)
}
if len(obs) != 0 || len(f.calls) != 0 {
t.Fatalf("observations=%d llm calls=%d, want 0/0 — own notes are not memory to evaluate", len(obs), len(f.calls))
}
}
// TestEvaluateEmptyArrayIsNotAnError — "nothing to say" is the expected outcome
// most of the time and must not be logged as a failure.
func TestEvaluateEmptyArrayIsNotAnError(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
now := refNow()
seedMemory(t, st, ctx, now)
ev := NewEvaluator(st, st, &fakeLLM{replies: []string{"[]"}}, Config{})
obs, err := ev.Evaluate(ctx, now)
if err != nil {
t.Fatalf("Evaluate: %v", err)
}
if len(obs) != 0 {
t.Fatalf("observations = %d, want 0", len(obs))
}
}
// TestEvaluateLLMErrorIsReported — a broken llama-server is an error the caller
// logs; it must not silently write anything.
func TestEvaluateLLMErrorIsReported(t *testing.T) {
st := newTestStore(t)
ctx := context.Background()
now := refNow()
seedMemory(t, st, ctx, now)
ev := NewEvaluator(st, st, &fakeLLM{err: errors.New("connection refused")}, Config{})
if _, err := ev.Evaluate(ctx, now); err == nil {
t.Fatal("want an error when the model is unreachable")
}
notes, err := st.RecentNotes(ctx, 50)
if err != nil {
t.Fatalf("RecentNotes: %v", err)
}
for _, n := range notes {
if n.Source == EvalNoteSource {
t.Fatalf("wrote a note despite an LLM failure: %q", n.Text)
}
}
}
// TestParseObservationsTolerantAndBounded — Thinking models wrap JSON in prose,
// and no reply may exceed MaxObservations even if the grammar is bypassed.
func TestParseObservationsTolerantAndBounded(t *testing.T) {
obs, err := parseObservations(`<think>hmm</think> вот: [{"observation":"a","confidence":0.9,"suggested_action":"note"}] всё`)
if err != nil {
t.Fatalf("parse: %v", err)
}
if len(obs) != 1 || obs[0].Text != "a" {
t.Fatalf("got %+v, want one observation 'a'", obs)
}
var b strings.Builder
b.WriteString("[")
for i := 0; i < MaxObservations+3; i++ {
if i > 0 {
b.WriteString(",")
}
b.WriteString(`{"observation":"x","confidence":0.5,"suggested_action":"note"}`)
}
b.WriteString("]")
obs, err = parseObservations(b.String())
if err != nil {
t.Fatalf("parse: %v", err)
}
if len(obs) != MaxObservations {
t.Fatalf("parsed %d observations, want the %d cap", len(obs), MaxObservations)
}
}