Add background memory evaluation, off unless configured (#248)
Ships the real, local, testable part of the memory-evaluation plan
(docs/plans/03-memory-evaluation.md): Maven reads back her own recent
memory on a slow ticker, asks the resident model what it notices, and
records the confident answers as notes.
internal/memeval — not internal/memory/eval.go as the plan says, because
internal/store imports internal/memory for the vector backend and an
evaluator has to read store.Fact/Note/Nudge, which would close the
cycle. Evaluate() gathers RecentFacts/RecentNotes/RecentNudges, prompts
under a GBNF grammar bounded to three {observation, confidence,
suggested_action} objects, drops anything under min_confidence,
deduplicates against what earlier runs wrote, and writes the rest as
notes with source infer:memory-eval. /dash already renders notes with
their source, so the output is visible with no UI change.
cmd/mavend/memoryeval.go drives it on its own goroutine and ticker, not
on the 60s tick: an evaluation is a multi-second round-trip on the same
llama-server that answers voice turns, and it runs hourly at most. The
memory_eval config block is absent by default and absence means the
goroutine does not exist. No llama-server phraser also means no loop —
there is no template fallback, because a "memory evaluation" assembled
from templates is a fixed sentence pretending to be an observation.
What it deliberately cannot do, since this is the feature most likely to
turn Maven into a nag:
- It cannot speak. No dispatcher reference, no channel, no nudge. An
observation is a thought she wrote down and he reads on /dash.
Announcing them is a separate decision with its own opt-in.
- It cannot act. suggested_action is recorded as text and interpreted
by nobody — no reminder, routine or fact is created from it.
- It says nothing about an empty store: no memory means no LLM call,
so there are no observations invented out of two facts.
- Its own notes are excluded from the next evaluation's input, and are
written with a nil embedding so they stay out of the recall pool.
The plan's remaining items (dispatching observations, an /eval IPC
method and trace view, RecentEvents) and the fact that output quality is
entirely unmeasured are written up at the bottom of the plan doc.
This commit is contained in:
@@ -169,6 +169,7 @@ func run(args []string) error {
|
||||
coreAPI ipc.CoreAPI
|
||||
eco *ecosystemWiring
|
||||
factWorker *factEnrichmentWorker
|
||||
evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default)
|
||||
)
|
||||
|
||||
if !locked {
|
||||
@@ -263,6 +264,7 @@ func run(args []string) error {
|
||||
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
||||
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals)
|
||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||
|
||||
coreAPI = &daemonAPI{
|
||||
CoreAPI: ipc.NewStoreAPI(st),
|
||||
@@ -440,6 +442,7 @@ func run(args []string) error {
|
||||
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
||||
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals)
|
||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||
|
||||
// Swap the CoreAPI from the locked placeholder to the real store adapter.
|
||||
newAPI := &daemonAPI{
|
||||
@@ -476,6 +479,13 @@ func run(args []string) error {
|
||||
factWorker.run(ctx)
|
||||
}()
|
||||
|
||||
// Start background memory evaluation (nil unless configured).
|
||||
if evalWorker != nil {
|
||||
go func() {
|
||||
evalWorker.run(ctx)
|
||||
}()
|
||||
}
|
||||
|
||||
dl.unlock()
|
||||
log.Printf("mavend: unlocked via passkey assertion")
|
||||
return nil
|
||||
@@ -514,6 +524,13 @@ func run(args []string) error {
|
||||
defer wg.Done()
|
||||
factWorker.run(ctx)
|
||||
}()
|
||||
if evalWorker != nil {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
evalWorker.run(ctx)
|
||||
}()
|
||||
}
|
||||
}
|
||||
|
||||
<-ctx.Done()
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
// mavend/memoryeval.go — the driver for background memory evaluation
|
||||
// (Vikunja #248). The evaluator itself is pure-ish and lives in
|
||||
// internal/memeval; this is the one impure part: a ticker, the store, and the
|
||||
// resident model's base URL.
|
||||
//
|
||||
// It is its own goroutine and NOT a step on the main tick, deliberately. The
|
||||
// tick runs every 60s and has a delivery deadline behind it; an evaluation is
|
||||
// a multi-second LLM round-trip on the same llama-server that answers voice
|
||||
// turns, and it happens hourly at most. Bolting it onto the tick would make
|
||||
// every hour's tick the slow one for no benefit.
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/config"
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/memeval"
|
||||
"github.com/kami/maven/internal/phraser"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
// memoryEvalWorker — ticker + evaluator.
|
||||
type memoryEvalWorker struct {
|
||||
eval *memeval.Evaluator
|
||||
interval time.Duration
|
||||
}
|
||||
|
||||
// newMemoryEvalWorker wires the evaluation loop, or returns nil when it should
|
||||
// not run at all. nil is the normal case and every caller must handle it:
|
||||
//
|
||||
// - no memory_eval config block ⇒ off (a capability is off unless configured);
|
||||
// - no LLM phraser ⇒ nothing to evaluate with. There is no template fallback
|
||||
// here on purpose: a "memory evaluation" assembled from string templates
|
||||
// would be a fixed sentence pretending to be an observation.
|
||||
func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Config) *memoryEvalWorker {
|
||||
if cfg.MemoryEval == nil {
|
||||
return nil
|
||||
}
|
||||
lp, ok := phr.(*phraser.LLMPhraser)
|
||||
if !ok {
|
||||
log.Printf("memory eval: configured but no llama-server phraser — evaluation disabled")
|
||||
return nil
|
||||
}
|
||||
interval := time.Duration(cfg.MemoryEval.Interval)
|
||||
if interval <= 0 {
|
||||
interval = config.DefaultMemoryEvalInterval
|
||||
}
|
||||
// A generous per-request timeout: this is a long prompt to a Thinking model
|
||||
// and nobody is waiting on the answer.
|
||||
client := llm.New(lp.BaseURL(), 5*time.Minute)
|
||||
ev := memeval.NewEvaluator(st, st, client, memeval.Config{
|
||||
MaxItems: cfg.MemoryEval.MaxItems,
|
||||
MinConfidence: cfg.MemoryEval.MinConfidence,
|
||||
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||
})
|
||||
log.Printf("memory eval: enabled, every %s", interval)
|
||||
return &memoryEvalWorker{eval: ev, interval: interval}
|
||||
}
|
||||
|
||||
// run evaluates every interval until ctx is canceled.
|
||||
//
|
||||
// The first evaluation waits a full interval rather than firing at startup, the
|
||||
// opposite of the tick loop's cold-start behaviour. A tick that fires late is a
|
||||
// nudge that arrives late; an evaluation that fires late is nothing at all, and
|
||||
// the alternative is a heavy LLM call competing with startup — including with
|
||||
// the first voice turn after a restart.
|
||||
func (w *memoryEvalWorker) run(ctx context.Context) {
|
||||
ticker := time.NewTicker(w.interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case now := <-ticker.C:
|
||||
obs, err := w.eval.Evaluate(ctx, now)
|
||||
if err != nil {
|
||||
log.Printf("memory eval: %v", err)
|
||||
continue
|
||||
}
|
||||
for _, o := range obs {
|
||||
log.Printf("memory eval: noted (%.2f, %s): %s", o.Conf, o.Action, o.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -25,3 +25,46 @@
|
||||
6. Add `/eval` API method to `ipc.CoreAPI` (or reuse `Chat` with system context) so mavweb can show evaluation history
|
||||
7. Add `memory_eval` block to `deploy/mavend.json`
|
||||
8. Test with synthetic store state — verify observations match expected patterns
|
||||
|
||||
---
|
||||
|
||||
## Status 2026-08-01 — foundation shipped (Vikunja #248)
|
||||
|
||||
**Shipped:** `internal/memeval` (not `internal/memory/eval.go` — `internal/store`
|
||||
imports `internal/memory` for the vector backend, so an evaluator that reads
|
||||
`store.Fact` there would close an import cycle). `Evaluator.Evaluate` reads
|
||||
`RecentFacts` / `RecentNotes` / `RecentNudges`, prompts the resident model under
|
||||
a GBNF grammar for at most three `{observation, confidence, suggested_action}`
|
||||
objects, drops anything under `min_confidence`, deduplicates against what earlier
|
||||
evaluations wrote, and records the rest as notes with source `infer:memory-eval`.
|
||||
Driver: `cmd/mavend/memoryeval.go`, its own goroutine on its own ticker. Config:
|
||||
the `memory_eval` block — **absent ⇒ the loop does not run**. Visibility: `/dash`
|
||||
already renders notes with their source, so evaluation output is visible with no
|
||||
UI change.
|
||||
|
||||
**Deliberately not shipped — this is policy, not an unfinished edge:**
|
||||
|
||||
- *Dispatching observations as care nudges (plan step 4).* An hourly LLM loop
|
||||
with permission to speak is a machine for generating interruptions, and the
|
||||
content is model-generated text about his own life. The evaluator has no
|
||||
dispatcher reference at all, so it cannot reach a channel by accident. Wiring
|
||||
it to `delivery.Dispatcher` is a separate decision with its own opt-in.
|
||||
- *Acting on `suggested_action`.* It is recorded inside the note text and
|
||||
interpreted by nobody. No reminder, routine or fact is created.
|
||||
- *Writing observation embeddings.* Notes are written with a nil embedding, so
|
||||
they stay out of the RAG recall pool. Feeding generated text back into the pool
|
||||
it came from is how a small model starts citing its own guesses as evidence.
|
||||
|
||||
**Deferred, wants a decision or another capability:**
|
||||
|
||||
- *Plan step 6, the `/eval` IPC method and an evaluation-history view.* `/dash`
|
||||
covers reading the output; a dedicated trace surface is worth building once
|
||||
there is real output to look at, and it should probably show the prompt too.
|
||||
- *`RecentEvents`.* The plan lists it; the evaluator reads facts, notes and
|
||||
nudges. Detected action/object events already drive pattern proposals (#43), and
|
||||
duplicating them here would mostly re-derive that.
|
||||
- *Output quality is unmeasured.* There is no fixture for "did she notice
|
||||
something true". The tests cover the machinery — empty store, confidence floor,
|
||||
dedupe, own-notes exclusion, error handling — not the observations. Until
|
||||
someone reads a week of real output on `/dash`, treat the wording and the
|
||||
`min_confidence` default as unvalidated.
|
||||
|
||||
@@ -146,6 +146,10 @@ type Config struct {
|
||||
// PatternProposalConfig.
|
||||
PatternProposals *PatternProposalConfig `json:"pattern_proposals,omitempty"`
|
||||
|
||||
// MemoryEval — background memory evaluation (internal/memeval). nil /
|
||||
// absent ⇒ no evaluation loop at all. See MemoryEvalConfig.
|
||||
MemoryEval *MemoryEvalConfig `json:"memory_eval,omitempty"`
|
||||
|
||||
// Praxis — the ecosystem attention-state service. When configured, maven
|
||||
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
|
||||
// Maven never touches Praxis's database directly (ecosystem invariant: no
|
||||
@@ -392,6 +396,28 @@ func (p *PatternProposalConfig) AnnounceProposals() bool {
|
||||
return p != nil && p.Notify
|
||||
}
|
||||
|
||||
// MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248).
|
||||
// Absent ⇒ off, like every other capability that costs something the owner did
|
||||
// not ask for. Each evaluation is a full LLM round-trip on the one resident
|
||||
// model, which is the same model answering him; running it hourly by default
|
||||
// would put a multi-second stall in front of an occasional voice turn for a
|
||||
// feature he may not want.
|
||||
//
|
||||
// The loop only ever writes notes (source infer:memory-eval, visible on
|
||||
// /dash). It cannot speak — see internal/memeval.
|
||||
type MemoryEvalConfig struct {
|
||||
// Interval — how often to evaluate. 0 ⇒ DefaultMemoryEvalInterval.
|
||||
Interval Duration `json:"interval,omitempty"`
|
||||
|
||||
// MaxItems — recent facts / notes / nudges fed into one evaluation.
|
||||
// 0 ⇒ memeval.DefaultMaxItems.
|
||||
MaxItems int `json:"max_items,omitempty"`
|
||||
|
||||
// MinConfidence — observations the model scores below this are dropped.
|
||||
// 0 ⇒ memeval.DefaultMinConfidence.
|
||||
MinConfidence float64 `json:"min_confidence,omitempty"`
|
||||
}
|
||||
|
||||
// PhraserConfig — the LLM-backed phraser seam. The daemon spawns llama-server
|
||||
// as a managed subprocess and sends chat-completion requests to phrase nudge
|
||||
// and reminder messages. nil ⇒ the template-based Stub is used instead.
|
||||
@@ -493,6 +519,10 @@ const (
|
||||
// most. A proposal is never urgent; if two patterns surface in the same
|
||||
// hour, the second one waits, and the /routines page has it either way.
|
||||
DefaultProposalCooldown = 24 * time.Hour
|
||||
|
||||
// DefaultMemoryEvalInterval — the plan's cadence (1h) for the memory
|
||||
// evaluation loop, applied only when the block is present at all.
|
||||
DefaultMemoryEvalInterval = time.Hour
|
||||
)
|
||||
|
||||
// Load reads the JSON config at path and applies defaults. A missing file is
|
||||
@@ -571,6 +601,12 @@ func (c *Config) applyDefaults() {
|
||||
c.PatternProposals.Cooldown = Duration(DefaultProposalCooldown)
|
||||
}
|
||||
|
||||
// Same rule: absent stays nil (⇒ no evaluation loop), present gets defaults
|
||||
// so `{}` is a valid "on with the plan's cadence".
|
||||
if c.MemoryEval != nil && c.MemoryEval.Interval <= 0 {
|
||||
c.MemoryEval.Interval = Duration(DefaultMemoryEvalInterval)
|
||||
}
|
||||
|
||||
if c.Voice != nil {
|
||||
if c.Voice.RouterThreshold <= 0 {
|
||||
c.Voice.RouterThreshold = DefaultRouterThreshold
|
||||
|
||||
@@ -243,3 +243,53 @@ func TestDurationRoundTrip(t *testing.T) {
|
||||
t.Errorf("round-trip = %v, want %v", d2, d)
|
||||
}
|
||||
}
|
||||
|
||||
// Both new opt-in capabilities follow the same rule: absent block ⇒ nil ⇒ the
|
||||
// behaviour does not exist. Presence is the enable act, so a bare `{}` block is
|
||||
// valid and gets the defaults filled in.
|
||||
func TestOptInBlocksAbsentStayNil(t *testing.T) {
|
||||
c, err := Load(writeConfig(t, `{}`))
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
if c.PatternProposals != nil {
|
||||
t.Errorf("pattern_proposals absent but got %+v", c.PatternProposals)
|
||||
}
|
||||
if c.PatternProposals.AnnounceProposals() {
|
||||
t.Error("AnnounceProposals() true with no config block")
|
||||
}
|
||||
if c.MemoryEval != nil {
|
||||
t.Errorf("memory_eval absent but got %+v", c.MemoryEval)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOptInBlocksGetDefaultsWhenPresent(t *testing.T) {
|
||||
c, err := Load(writeConfig(t, `{"pattern_proposals":{"notify":true},"memory_eval":{}}`))
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
if !c.PatternProposals.AnnounceProposals() {
|
||||
t.Error("notify:true did not enable announcements")
|
||||
}
|
||||
if time.Duration(c.PatternProposals.Cooldown) != DefaultProposalCooldown {
|
||||
t.Errorf("proposal cooldown = %v, want %v", c.PatternProposals.Cooldown, DefaultProposalCooldown)
|
||||
}
|
||||
if time.Duration(c.MemoryEval.Interval) != DefaultMemoryEvalInterval {
|
||||
t.Errorf("memory eval interval = %v, want %v", c.MemoryEval.Interval, DefaultMemoryEvalInterval)
|
||||
}
|
||||
}
|
||||
|
||||
// Notify is off even when the block exists — the block is where you tune it,
|
||||
// notify:true is the act that lets her speak.
|
||||
func TestPatternProposalNotifyDefaultsOff(t *testing.T) {
|
||||
c, err := Load(writeConfig(t, `{"pattern_proposals":{"cooldown":"6h"}}`))
|
||||
if err != nil {
|
||||
t.Fatalf("Load: %v", err)
|
||||
}
|
||||
if c.PatternProposals.AnnounceProposals() {
|
||||
t.Error("notify defaulted to on")
|
||||
}
|
||||
if time.Duration(c.PatternProposals.Cooldown) != 6*time.Hour {
|
||||
t.Errorf("cooldown = %v, want 6h", c.PatternProposals.Cooldown)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,357 @@
|
||||
// Package memeval is background memory evaluation (Vikunja #248,
|
||||
// docs/plans/03-memory-evaluation.md).
|
||||
//
|
||||
// It lives beside internal/memory rather than inside it because
|
||||
// internal/store imports internal/memory for the vector-store backend, and an
|
||||
// evaluator has to read store.Fact / store.Note / store.Nudge — putting it in
|
||||
// internal/memory would close that import cycle.
|
||||
//
|
||||
// Every so often Maven reads back her own recent memory — facts, notes, the
|
||||
// nudges she sent — and asks the resident model what it notices: a habit that
|
||||
// stopped, a gap, something worth saying later. What comes back is written as
|
||||
// notes with source EvalNoteSource and nothing else happens. That restraint is
|
||||
// the design, not an unfinished edge:
|
||||
//
|
||||
// - She does not speak here. There is no dispatcher, no channel, no nudge.
|
||||
// An observation is a thought she wrote down; he reads it on /dash when he
|
||||
// wants to. "Not a nag, not autonomous" (CLAUDE.md) is easy to violate with
|
||||
// exactly this feature — an hourly loop with an LLM in it and permission to
|
||||
// talk is a machine for generating interruptions — so the loop has no way
|
||||
// to reach him at all. Turning observations into nudges is a separate
|
||||
// decision with a separate opt-in, and it is deliberately NOT in this file.
|
||||
// - She does not act. No reminder is created, no routine proposed, no fact
|
||||
// written. The model's suggested_action is recorded as text inside the note
|
||||
// and interpreted by nobody.
|
||||
// - She says nothing about an empty store. No memory ⇒ no LLM call ⇒ no
|
||||
// "observations" invented out of two facts. A 1.7B asked to find a pattern
|
||||
// will always find one; the defence is not asking.
|
||||
//
|
||||
// Everything the evaluator writes is attributable: source is EvalNoteSource, so
|
||||
// an inferred observation can never be mistaken for something he said, and the
|
||||
// whole batch is one SQL delete away if the output turns out to be noise.
|
||||
package memeval
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/persona"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
// EvalNoteSource — the source stamped on every note the evaluator writes.
|
||||
// Same infer:* convention as the rest of the derived facts.
|
||||
const EvalNoteSource = "infer:memory-eval"
|
||||
|
||||
// DefaultMinConfidence — an observation below this is dropped. The model is
|
||||
// asked for its own confidence and small models are badly calibrated, so this
|
||||
// is a coarse filter, not a probability: it exists to throw away the guesses
|
||||
// the model itself hedged on.
|
||||
const DefaultMinConfidence = 0.7
|
||||
|
||||
// DefaultMaxItems — how much recent memory goes into one evaluation, per
|
||||
// store. 30 facts + 30 notes + 30 nudges is a few thousand tokens of the 4096
|
||||
// context the resident Thinking model runs with, which leaves room for its
|
||||
// reasoning tokens. Raising this trades reasoning room for history.
|
||||
const DefaultMaxItems = 30
|
||||
|
||||
// MaxObservations — the model may return at most this many observations per
|
||||
// evaluation, enforced by the grammar. A cap here is also a noise cap: an
|
||||
// evaluation that "notices" ten things has noticed nothing.
|
||||
const MaxObservations = 3
|
||||
|
||||
// Observation — one thing the evaluator noticed.
|
||||
type Observation struct {
|
||||
Text string `json:"observation"`
|
||||
Conf float64 `json:"confidence"`
|
||||
// Action — what the model thinks should happen with this. Recorded, never
|
||||
// executed: see the file comment. One of "note", "propose", "notify".
|
||||
Action string `json:"suggested_action"`
|
||||
}
|
||||
|
||||
// Completer — the llama-server seam, same shape router.Completer uses so the
|
||||
// one resident model serves this caller too.
|
||||
type Completer interface {
|
||||
Complete(ctx context.Context, r llm.Req) (string, error)
|
||||
}
|
||||
|
||||
// Reader — the slice of the store an evaluation reads. Narrow on purpose: the
|
||||
// evaluator gets recent memory and nothing else. No entity graph, no presence,
|
||||
// no config facts.
|
||||
type Reader interface {
|
||||
RecentFacts(ctx context.Context, n int) ([]store.Fact, error)
|
||||
RecentNotes(ctx context.Context, n int) ([]store.Note, error)
|
||||
RecentNudges(ctx context.Context, n int) ([]store.Nudge, error)
|
||||
}
|
||||
|
||||
// NoteWriter — where observations land. Embeddings are passed nil: an
|
||||
// observation is written for a human to read on /dash, not to be recalled by
|
||||
// similarity. Feeding LLM-generated text back into the RAG pool it was
|
||||
// generated from is how a small model starts citing its own guesses as
|
||||
// evidence.
|
||||
type NoteWriter interface {
|
||||
WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error)
|
||||
}
|
||||
|
||||
// Config — evaluator tuning. Zero values are replaced by the Default*
|
||||
// constants, so the zero Config is the sane one.
|
||||
type Config struct {
|
||||
MaxItems int
|
||||
MinConfidence float64
|
||||
|
||||
// ContextBlock — the shared persona block (internal/persona), re-evaluated
|
||||
// per call so the clock in it is current. Prepended to the system prompt so
|
||||
// observations come out in Maven's voice: feminine self-reference, informal
|
||||
// "ты". nil is allowed; the base prompt still carries the address rules.
|
||||
ContextBlock func() string
|
||||
}
|
||||
|
||||
// Evaluator reads recent memory and records what the model notices.
|
||||
type Evaluator struct {
|
||||
read Reader
|
||||
write NoteWriter
|
||||
llm Completer
|
||||
cfg Config
|
||||
}
|
||||
|
||||
func NewEvaluator(r Reader, w NoteWriter, c Completer, cfg Config) *Evaluator {
|
||||
if cfg.MaxItems <= 0 {
|
||||
cfg.MaxItems = DefaultMaxItems
|
||||
}
|
||||
if cfg.MinConfidence <= 0 {
|
||||
cfg.MinConfidence = DefaultMinConfidence
|
||||
}
|
||||
return &Evaluator{read: r, write: w, llm: c, cfg: cfg}
|
||||
}
|
||||
|
||||
// evalGrammar — GBNF pinning the reply to a bounded JSON array of fixed-shape
|
||||
// observations. Same reasoning as the router's routeGrammar: the enum and the
|
||||
// length bound are what stop a small model from drifting into free text or
|
||||
// filling the token budget with one repeated field.
|
||||
const evalGrammar = `
|
||||
root ::= "[" ws (obs ("," ws obs){0,2})? ws "]"
|
||||
obs ::= "{" ws "\"observation\"" ws ":" ws text "," ws "\"confidence\"" ws ":" ws conf "," ws "\"suggested_action\"" ws ":" ws act ws "}"
|
||||
text ::= "\"" ([^"\\] | "\\" .){1,200} "\""
|
||||
conf ::= "0" "." [0-9]{1,2} | "1" ("." "0")?
|
||||
act ::= "\"note\"" | "\"propose\"" | "\"notify\""
|
||||
ws ::= [ \t\n]*
|
||||
`
|
||||
|
||||
// evalSystem — the evaluation prompt. Two things it insists on, both learned
|
||||
// from the phraser: state the observation as something she noticed rather than
|
||||
// an instruction, and say nothing when there is nothing (the model is given an
|
||||
// explicit way to return an empty array, because a model with no exit returns
|
||||
// filler).
|
||||
const evalSystem = `Ты просматриваешь свою собственную память: недавние факты, заметки и напоминания, которые ты отправляла.
|
||||
Найди то, что действительно заметно: привычка, которая прервалась; пробел в записях; повторяющаяся закономерность.
|
||||
|
||||
Правила:
|
||||
- Отвечай ТОЛЬКО массивом JSON. Каждый элемент: {"observation": "...", "confidence": 0.0-1.0, "suggested_action": "note"|"propose"|"notify"}.
|
||||
- observation — короткая фраза по-русски о том, что ты заметила. О себе — в женском роде ("я заметила"). К нему — на "ты".
|
||||
- Не выдумывай. Если в памяти нет ничего заметного, верни пустой массив [].
|
||||
- Не давай советов и не приказывай. Ты замечаешь, а не требуешь.
|
||||
- confidence — насколько ты уверена, что это настоящая закономерность, а не совпадение.
|
||||
- Максимум три наблюдения. Лучше одно точное, чем три общих.`
|
||||
|
||||
// Evaluate runs one evaluation and returns the observations it recorded.
|
||||
//
|
||||
// Returns (nil, nil) — not an error — for every ordinary "nothing to say"
|
||||
// outcome: an empty store, an empty array from the model, everything below the
|
||||
// confidence floor, or every observation already recorded earlier. Only a real
|
||||
// read/LLM/write failure is an error, and the caller (a background ticker) logs
|
||||
// it and waits for the next interval.
|
||||
func (e *Evaluator) Evaluate(ctx context.Context, now time.Time) ([]Observation, error) {
|
||||
snap, err := e.snapshot(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if snap == "" {
|
||||
return nil, nil // nothing recorded ⇒ nothing to notice, and no LLM call
|
||||
}
|
||||
|
||||
raw, err := e.llm.Complete(ctx, llm.Req{
|
||||
System: persona.Prepend(e.cfg.ContextBlock, evalSystem),
|
||||
User: snap,
|
||||
Grammar: evalGrammar,
|
||||
MaxTokens: 512,
|
||||
RepeatPenalty: 1.1,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("memory eval: complete: %w", err)
|
||||
}
|
||||
obs, err := parseObservations(raw)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("memory eval: parse %q: %w", truncate(raw, 120), err)
|
||||
}
|
||||
|
||||
// Dedupe against what earlier evaluations already wrote. Without this an
|
||||
// hourly loop over a slowly-changing store writes the same sentence every
|
||||
// hour until /dash is nothing but the evaluator talking to itself.
|
||||
seen, err := e.recordedTexts(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
var kept []Observation
|
||||
for _, o := range obs {
|
||||
o.Text = strings.TrimSpace(o.Text)
|
||||
if o.Text == "" || o.Conf < e.cfg.MinConfidence {
|
||||
continue
|
||||
}
|
||||
norm := normalizeObservation(o.Text)
|
||||
if seen[norm] {
|
||||
continue
|
||||
}
|
||||
seen[norm] = true
|
||||
if _, err := e.write.WriteNote(ctx, now, formatNote(o), nil, EvalNoteSource); err != nil {
|
||||
return kept, fmt.Errorf("memory eval: write note: %w", err)
|
||||
}
|
||||
kept = append(kept, o)
|
||||
}
|
||||
return kept, nil
|
||||
}
|
||||
|
||||
// formatNote — the stored text. The suggested action is kept as a visible
|
||||
// suffix rather than a column: it is the model's opinion about what to do next,
|
||||
// and the only consumer is a human reading /dash.
|
||||
func formatNote(o Observation) string {
|
||||
if o.Action == "" {
|
||||
return o.Text
|
||||
}
|
||||
return fmt.Sprintf("%s [%s]", o.Text, o.Action)
|
||||
}
|
||||
|
||||
// recordedTexts — the normalized text of every observation earlier evaluations
|
||||
// wrote, for dedupe. Reads a wider window than MaxItems because the point is to
|
||||
// remember saying it, not to summarize it.
|
||||
func (e *Evaluator) recordedTexts(ctx context.Context) (map[string]bool, error) {
|
||||
notes, err := e.read.RecentNotes(ctx, 200)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("memory eval: recent notes: %w", err)
|
||||
}
|
||||
seen := make(map[string]bool, len(notes))
|
||||
for _, n := range notes {
|
||||
if n.Source != EvalNoteSource {
|
||||
continue
|
||||
}
|
||||
text := n.Text
|
||||
// Strip the "[action]" suffix formatNote appended.
|
||||
if i := strings.LastIndex(text, " ["); i > 0 && strings.HasSuffix(text, "]") {
|
||||
text = text[:i]
|
||||
}
|
||||
seen[normalizeObservation(text)] = true
|
||||
}
|
||||
return seen, nil
|
||||
}
|
||||
|
||||
// normalizeObservation — dedupe key. Case- and whitespace-insensitive, which
|
||||
// catches the realistic repeat (the model re-emitting the same sentence with a
|
||||
// different comma) without pretending to do semantic dedupe.
|
||||
func normalizeObservation(s string) string {
|
||||
return strings.Join(strings.Fields(strings.ToLower(s)), " ")
|
||||
}
|
||||
|
||||
// snapshot renders recent memory as the user turn. Returns "" when there is
|
||||
// nothing in any store — the caller treats that as "do not ask the model".
|
||||
//
|
||||
// Notes written by earlier evaluations are excluded. Feeding her own
|
||||
// observations back in is how "я заметила, что ты не записывал еду" becomes
|
||||
// evidence for noticing it again, three evaluations deep.
|
||||
func (e *Evaluator) snapshot(ctx context.Context) (string, error) {
|
||||
n := e.cfg.MaxItems
|
||||
facts, err := e.read.RecentFacts(ctx, n)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("memory eval: recent facts: %w", err)
|
||||
}
|
||||
notes, err := e.read.RecentNotes(ctx, n)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("memory eval: recent notes: %w", err)
|
||||
}
|
||||
nudges, err := e.read.RecentNudges(ctx, n)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("memory eval: recent nudges: %w", err)
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
wrote := false
|
||||
if len(facts) > 0 {
|
||||
b.WriteString("Факты:\n")
|
||||
for _, f := range facts {
|
||||
fmt.Fprintf(&b, "- %s %s=%s (%s)\n", f.Ts.Format("2006-01-02 15:04"), f.Key, truncate(f.Value, 80), f.Source)
|
||||
wrote = true
|
||||
}
|
||||
}
|
||||
own := 0
|
||||
var noteLines []string
|
||||
for _, nt := range notes {
|
||||
if nt.Source == EvalNoteSource {
|
||||
own++
|
||||
continue
|
||||
}
|
||||
noteLines = append(noteLines, fmt.Sprintf("- %s %s\n", nt.Ts.Format("2006-01-02 15:04"), truncate(nt.Text, 160)))
|
||||
}
|
||||
if len(noteLines) > 0 {
|
||||
b.WriteString("\nЗаметки:\n")
|
||||
for _, l := range noteLines {
|
||||
b.WriteString(l)
|
||||
wrote = true
|
||||
}
|
||||
}
|
||||
if len(nudges) > 0 {
|
||||
b.WriteString("\nНапоминания, которые ты отправляла:\n")
|
||||
for _, nd := range nudges {
|
||||
outcome := nd.Outcome
|
||||
if outcome == "" {
|
||||
outcome = "?"
|
||||
}
|
||||
fmt.Fprintf(&b, "- %s %s → %s (%s)\n", nd.Ts.Format("2006-01-02 15:04"), nd.Rule, outcome, nd.Channel)
|
||||
wrote = true
|
||||
}
|
||||
}
|
||||
if !wrote {
|
||||
// Only her own past observations, or nothing at all. Either way there is
|
||||
// no new memory to evaluate.
|
||||
return "", nil
|
||||
}
|
||||
b.WriteString("\nЧто ты замечаешь?")
|
||||
return b.String(), nil
|
||||
}
|
||||
|
||||
// parseObservations reads the model's array. Tolerates the leading/trailing
|
||||
// prose a Thinking model sometimes emits around JSON by taking the outermost
|
||||
// bracketed span, the same tolerance the router's parser has.
|
||||
func parseObservations(raw string) ([]Observation, error) {
|
||||
s := strings.TrimSpace(raw)
|
||||
if i := strings.Index(s, "["); i >= 0 {
|
||||
if j := strings.LastIndex(s, "]"); j > i {
|
||||
s = s[i : j+1]
|
||||
}
|
||||
}
|
||||
if s == "" {
|
||||
return nil, nil
|
||||
}
|
||||
var obs []Observation
|
||||
if err := json.Unmarshal([]byte(s), &obs); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if len(obs) > MaxObservations {
|
||||
// The grammar bounds this; a grammar-less server or a future prompt
|
||||
// change must not be able to flood /dash.
|
||||
sort.SliceStable(obs, func(i, j int) bool { return obs[i].Conf > obs[j].Conf })
|
||||
obs = obs[:MaxObservations]
|
||||
}
|
||||
return obs, nil
|
||||
}
|
||||
|
||||
func truncate(s string, n int) string {
|
||||
r := []rune(s)
|
||||
if len(r) <= n {
|
||||
return s
|
||||
}
|
||||
return string(r[:n]) + "…"
|
||||
}
|
||||
@@ -0,0 +1,271 @@
|
||||
package memeval
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/llm"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
// fakeLLM — canned replies, one per call, and a record of what it was asked.
|
||||
type fakeLLM struct {
|
||||
replies []string
|
||||
calls []llm.Req
|
||||
err error
|
||||
}
|
||||
|
||||
func (f *fakeLLM) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||
f.calls = append(f.calls, r)
|
||||
if f.err != nil {
|
||||
return "", f.err
|
||||
}
|
||||
if len(f.replies) == 0 {
|
||||
return "[]", nil
|
||||
}
|
||||
out := f.replies[0]
|
||||
f.replies = f.replies[1:]
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func newTestStore(t *testing.T) *store.Store {
|
||||
t.Helper()
|
||||
st, err := store.Open(context.Background(), filepath.Join(t.TempDir(), "memeval_test.db"))
|
||||
if err != nil {
|
||||
t.Fatalf("store.Open: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { _ = st.Close() })
|
||||
return st
|
||||
}
|
||||
|
||||
func refNow() time.Time { return time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC) }
|
||||
|
||||
// seedMemory writes a little of everything the evaluator reads.
|
||||
func seedMemory(t *testing.T, st *store.Store, ctx context.Context, now time.Time) {
|
||||
t.Helper()
|
||||
for i := 0; i < 3; i++ {
|
||||
ts := now.Add(-time.Duration(i+1) * 24 * time.Hour)
|
||||
if _, err := st.WriteFact(ctx, ts, store.KindSelf, "water_ml", "500", "tap:desk", 1.0, sql.NullInt64{}); err != nil {
|
||||
t.Fatalf("write fact: %v", err)
|
||||
}
|
||||
}
|
||||
if _, err := st.WriteNote(ctx, now.Add(-2*time.Hour), "купить корм для кота", nil, "tap:voice"); err != nil {
|
||||
t.Fatalf("write note: %v", err)
|
||||
}
|
||||
if _, err := st.RecordNudge(ctx, "water", "voice", "пора выпить воды", now.Add(-time.Hour)); err != nil {
|
||||
t.Fatalf("record nudge: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateEmptyStoreAsksNothing — the "shuts up when uncertain" floor. An
|
||||
// empty store must not even reach the model: a small model asked to find a
|
||||
// pattern in nothing will invent one.
|
||||
func TestEvaluateEmptyStoreAsksNothing(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
f := &fakeLLM{}
|
||||
ev := NewEvaluator(st, st, f, Config{})
|
||||
|
||||
obs, err := ev.Evaluate(ctx, refNow())
|
||||
if err != nil {
|
||||
t.Fatalf("Evaluate: %v", err)
|
||||
}
|
||||
if len(obs) != 0 {
|
||||
t.Fatalf("observations on an empty store = %d, want 0", len(obs))
|
||||
}
|
||||
if len(f.calls) != 0 {
|
||||
t.Fatalf("LLM called %d times on an empty store, want 0", len(f.calls))
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateWritesHighConfidenceObservations — the happy path. Confident
|
||||
// observations are written as notes stamped infer:memory-eval, and the low
|
||||
// ones are dropped.
|
||||
func TestEvaluateWritesHighConfidenceObservations(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
now := refNow()
|
||||
seedMemory(t, st, ctx, now)
|
||||
|
||||
f := &fakeLLM{replies: []string{`[
|
||||
{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"notify"},
|
||||
{"observation":"может быть, ты стал меньше пить воды","confidence":0.3,"suggested_action":"note"}
|
||||
]`}}
|
||||
ev := NewEvaluator(st, st, f, Config{})
|
||||
|
||||
obs, err := ev.Evaluate(ctx, now)
|
||||
if err != nil {
|
||||
t.Fatalf("Evaluate: %v", err)
|
||||
}
|
||||
if len(obs) != 1 {
|
||||
t.Fatalf("kept %d observations, want 1 (the 0.3 one is below the floor): %+v", len(obs), obs)
|
||||
}
|
||||
if obs[0].Text != "ты три дня не записывал еду" {
|
||||
t.Errorf("kept the wrong observation: %q", obs[0].Text)
|
||||
}
|
||||
|
||||
notes, err := st.RecentNotes(ctx, 50)
|
||||
if err != nil {
|
||||
t.Fatalf("RecentNotes: %v", err)
|
||||
}
|
||||
var written []store.Note
|
||||
for _, n := range notes {
|
||||
if n.Source == EvalNoteSource {
|
||||
written = append(written, n)
|
||||
}
|
||||
}
|
||||
if len(written) != 1 {
|
||||
t.Fatalf("notes with source %s = %d, want 1", EvalNoteSource, len(written))
|
||||
}
|
||||
if !strings.Contains(written[0].Text, "ты три дня не записывал еду") {
|
||||
t.Errorf("note text = %q", written[0].Text)
|
||||
}
|
||||
if !strings.Contains(written[0].Text, "[notify]") {
|
||||
t.Errorf("note text = %q, want the suggested action recorded", written[0].Text)
|
||||
}
|
||||
|
||||
// The prompt must carry the memory it is evaluating, and must not carry a
|
||||
// grammar-free request.
|
||||
if len(f.calls) != 1 {
|
||||
t.Fatalf("LLM calls = %d, want 1", len(f.calls))
|
||||
}
|
||||
if !strings.Contains(f.calls[0].User, "water_ml") {
|
||||
t.Errorf("prompt does not mention the seeded facts:\n%s", f.calls[0].User)
|
||||
}
|
||||
if f.calls[0].Grammar == "" {
|
||||
t.Error("evaluation ran without a grammar")
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateDeduplicatesAcrossRuns — the failure mode that would make this
|
||||
// feature unusable: an hourly loop over a store that barely changes writing the
|
||||
// same sentence every hour until /dash is nothing but the evaluator.
|
||||
func TestEvaluateDeduplicatesAcrossRuns(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
now := refNow()
|
||||
seedMemory(t, st, ctx, now)
|
||||
|
||||
same := `[{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"note"}]`
|
||||
spaced := `[{"observation":"Ты три дня не записывал еду","confidence":0.95,"suggested_action":"note"}]`
|
||||
f := &fakeLLM{replies: []string{same, same, spaced}}
|
||||
ev := NewEvaluator(st, st, f, Config{})
|
||||
|
||||
for i := 0; i < 3; i++ {
|
||||
if _, err := ev.Evaluate(ctx, now.Add(time.Duration(i)*time.Hour)); err != nil {
|
||||
t.Fatalf("Evaluate %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
|
||||
notes, err := st.RecentNotes(ctx, 50)
|
||||
if err != nil {
|
||||
t.Fatalf("RecentNotes: %v", err)
|
||||
}
|
||||
n := 0
|
||||
for _, nt := range notes {
|
||||
if nt.Source == EvalNoteSource {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n != 1 {
|
||||
t.Fatalf("eval notes after three identical evaluations = %d, want 1", n)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateIgnoresOwnNotes — her own observations must not become input.
|
||||
// Otherwise "я заметила X" is evidence for noticing X again, three evaluations
|
||||
// deep. With nothing but eval notes in the store there is no new memory, so the
|
||||
// model is not asked at all.
|
||||
func TestEvaluateIgnoresOwnNotes(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
now := refNow()
|
||||
if _, err := st.WriteNote(ctx, now.Add(-time.Hour), "я заметила, что ты мало пьёшь [note]", nil, EvalNoteSource); err != nil {
|
||||
t.Fatalf("write note: %v", err)
|
||||
}
|
||||
|
||||
f := &fakeLLM{}
|
||||
ev := NewEvaluator(st, st, f, Config{})
|
||||
obs, err := ev.Evaluate(ctx, now)
|
||||
if err != nil {
|
||||
t.Fatalf("Evaluate: %v", err)
|
||||
}
|
||||
if len(obs) != 0 || len(f.calls) != 0 {
|
||||
t.Fatalf("observations=%d llm calls=%d, want 0/0 — own notes are not memory to evaluate", len(obs), len(f.calls))
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateEmptyArrayIsNotAnError — "nothing to say" is the expected outcome
|
||||
// most of the time and must not be logged as a failure.
|
||||
func TestEvaluateEmptyArrayIsNotAnError(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
now := refNow()
|
||||
seedMemory(t, st, ctx, now)
|
||||
|
||||
ev := NewEvaluator(st, st, &fakeLLM{replies: []string{"[]"}}, Config{})
|
||||
obs, err := ev.Evaluate(ctx, now)
|
||||
if err != nil {
|
||||
t.Fatalf("Evaluate: %v", err)
|
||||
}
|
||||
if len(obs) != 0 {
|
||||
t.Fatalf("observations = %d, want 0", len(obs))
|
||||
}
|
||||
}
|
||||
|
||||
// TestEvaluateLLMErrorIsReported — a broken llama-server is an error the caller
|
||||
// logs; it must not silently write anything.
|
||||
func TestEvaluateLLMErrorIsReported(t *testing.T) {
|
||||
st := newTestStore(t)
|
||||
ctx := context.Background()
|
||||
now := refNow()
|
||||
seedMemory(t, st, ctx, now)
|
||||
|
||||
ev := NewEvaluator(st, st, &fakeLLM{err: errors.New("connection refused")}, Config{})
|
||||
if _, err := ev.Evaluate(ctx, now); err == nil {
|
||||
t.Fatal("want an error when the model is unreachable")
|
||||
}
|
||||
notes, err := st.RecentNotes(ctx, 50)
|
||||
if err != nil {
|
||||
t.Fatalf("RecentNotes: %v", err)
|
||||
}
|
||||
for _, n := range notes {
|
||||
if n.Source == EvalNoteSource {
|
||||
t.Fatalf("wrote a note despite an LLM failure: %q", n.Text)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestParseObservationsTolerantAndBounded — Thinking models wrap JSON in prose,
|
||||
// and no reply may exceed MaxObservations even if the grammar is bypassed.
|
||||
func TestParseObservationsTolerantAndBounded(t *testing.T) {
|
||||
obs, err := parseObservations(`<think>hmm</think> вот: [{"observation":"a","confidence":0.9,"suggested_action":"note"}] всё`)
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if len(obs) != 1 || obs[0].Text != "a" {
|
||||
t.Fatalf("got %+v, want one observation 'a'", obs)
|
||||
}
|
||||
|
||||
var b strings.Builder
|
||||
b.WriteString("[")
|
||||
for i := 0; i < MaxObservations+3; i++ {
|
||||
if i > 0 {
|
||||
b.WriteString(",")
|
||||
}
|
||||
b.WriteString(`{"observation":"x","confidence":0.5,"suggested_action":"note"}`)
|
||||
}
|
||||
b.WriteString("]")
|
||||
obs, err = parseObservations(b.String())
|
||||
if err != nil {
|
||||
t.Fatalf("parse: %v", err)
|
||||
}
|
||||
if len(obs) != MaxObservations {
|
||||
t.Fatalf("parsed %d observations, want the %d cap", len(obs), MaxObservations)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user