// Package eval scores nudge phrasing — the sentences the operator actually // hears. It is the phrasing counterpart to internal/router/eval. // // Why a separate package from phraser: the fixture must be scorable by BOTH // phrasing paths (the deterministic Stub and the resident model) from outside // the phraser package, and a _test.go file inside phraser cannot be imported. // So the fixture is embedded here and the scorer takes a Nudger interface. // // Why deterministic checks and not model judgement: the resident model is a // 0.8B. It cannot grade its own tone. Every check in checks.go is a string or // length test that a human can read and disagree with. A score here is a claim // about measurable properties, not about whether a sentence is good. // // DESIGN.md § "Rules decide, LLM phrases" is why there is no send/veto signal // anywhere in this package: the rule already decided she speaks. The phraser // only words it, so a nudge the model refuses to write is a failure, never a // legitimate outcome. package eval import ( "context" _ "embed" "encoding/json" "fmt" "sort" "strings" "time" "github.com/kami/maven/internal/delivery" "github.com/kami/maven/internal/loop" "github.com/kami/maven/internal/store" ) //go:embed nudges_v1.json var fixtureJSON []byte // SchemaVersion — the version this package understands. const SchemaVersion = 1 // Case — one nudge situation, as a real tick would present it. The fields are // the (rule, severity, context) input DESIGN.md names, flattened to JSON. // // WantAny is the on-topic contract: at least one of these lowercased fragments // must appear in the message. A water nudge that never mentions water is a // failure however charming it reads. Fragments are stems ("вод") so declension // does not defeat the check, and they list both languages because the Stub is // still English (see the writeup). type Case struct { ID string `json:"id"` Rule string `json:"rule"` Severity int `json:"severity"` // SinceMinutes — age of the rule's own fact. 0 means "no such fact", which // is the branch where the phraser has no duration to name. SinceMinutes int `json:"since_minutes"` // FactKey/FactValue/FactSource — the aggregate fact behind the ops rules. // service_down phrasing reads the key for the service name. FactKey string `json:"fact_key,omitempty"` FactValue string `json:"fact_value,omitempty"` FactSource string `json:"fact_source,omitempty"` // QuietHours/CalendarBusy — the bad moments. The gate already let this // nudge through (ops outranks quiet hours), so the phrasing still has to be // short and plain rather than apologetic about the timing. QuietHours bool `json:"quiet_hours,omitempty"` CalendarBusy bool `json:"calendar_busy,omitempty"` WantAny []string `json:"want_any"` Tags []string `json:"tags,omitempty"` Note string `json:"note,omitempty"` } // Fixture — the versioned envelope. SchemaVersion gates the loader so an older // binary refuses a fixture it would misread instead of reporting a wrong score. type Fixture struct { SchemaVersion int `json:"schema_version"` Name string `json:"name"` ReferenceNow string `json:"reference_now"` Notes []string `json:"notes"` Cases []Case `json:"cases"` } // Load returns the embedded fixture. func Load() (Fixture, error) { var f Fixture if err := json.Unmarshal(fixtureJSON, &f); err != nil { return Fixture{}, fmt.Errorf("parse fixture: %w", err) } if f.SchemaVersion != SchemaVersion { return Fixture{}, fmt.Errorf("fixture schema_version %d, want %d", f.SchemaVersion, SchemaVersion) } if len(f.Cases) == 0 { return Fixture{}, fmt.Errorf("fixture has no cases") } return f, nil } // Now — the fixture's reference clock, so fact ages are reproducible. func (f Fixture) Now() (time.Time, error) { t, err := time.Parse(time.RFC3339, f.ReferenceNow) if err != nil { return time.Time{}, fmt.Errorf("parse reference_now %q: %w", f.ReferenceNow, err) } return t, nil } // Candidate rebuilds the loop.Candidate a tick would hand the phraser. func (c Case) Candidate(now time.Time) loop.Candidate { state := loop.State{ Now: now, Facts: map[string]store.Fact{}, QuietHours: c.QuietHours, CalendarBusy: c.CalendarBusy, } if c.SinceMinutes > 0 { state.Facts[c.Rule] = store.Fact{ Key: c.Rule, Ts: now.Add(-time.Duration(c.SinceMinutes) * time.Minute), } } if c.FactKey != "" { state.Facts[c.Rule] = store.Fact{ Key: c.FactKey, Value: c.FactValue, Source: c.FactSource, Ts: now.Add(-time.Duration(c.SinceMinutes) * time.Minute), } } sev := loop.Severity(c.Severity) return loop.Candidate{ Rule: loop.Rule{Name: c.Rule, Severity: sev}, Severity: sev, State: state, } } // Nudger — the one thing a phrasing path must do to be scorable. Both // *phraser.Stub and *phraser.LLMPhraser satisfy it. type Nudger interface { PhraseNudge(ctx context.Context, c loop.Candidate) (delivery.PhrasedNudge, error) } // Outcome — one scored case. Failed lists the check names that did not pass, // Reasons the human-readable detail. Failed is empty exactly when Pass is true. type Outcome struct { Case Case Body string Mood string Err error Latency time.Duration Pass bool Failed []string Reasons []string } // Report — the aggregate. ByCheck is the useful part: one composite percentage // hides which property broke, and tuning a prompt needs to know. type Report struct { Name string Total int Passed int Errors int ByCheck map[string]int ByRule map[string]TagStat Outcomes []Outcome P50 time.Duration P95 time.Duration Max time.Duration } // TagStat — passed/total for one slice of the fixture. type TagStat struct{ Passed, Total int } // Accuracy — fraction of cases that passed every check. func (r Report) Accuracy() float64 { if r.Total == 0 { return 0 } return float64(r.Passed) / float64(r.Total) } // Score runs every case through p and aggregates. A phrasing error scores as a // miss and is counted in Errors — "the model was down" and "the model wrote // something bad" are different numbers and a prompt change must not be able to // hide behind the first one. // // Latency is wall-clock per PhraseNudge call. On the CPU/iGPU target a nudge // the model takes a minute to word has already missed its moment. func Score(ctx context.Context, name string, p Nudger, f Fixture) (Report, error) { now, err := f.Now() if err != nil { return Report{}, err } rep := Report{ Name: name, Total: len(f.Cases), ByCheck: map[string]int{}, ByRule: map[string]TagStat{}, } for _, name := range CheckNames { rep.ByCheck[name] = 0 } lat := make([]time.Duration, 0, len(f.Cases)) for _, c := range f.Cases { start := time.Now() pn, err := p.PhraseNudge(ctx, c.Candidate(now)) o := Outcome{Case: c, Body: pn.Body, Mood: pn.Mood, Err: err, Latency: time.Since(start)} lat = append(lat, o.Latency) if err != nil { rep.Errors++ o.Failed = append(o.Failed, "call") o.Reasons = append(o.Reasons, fmt.Sprintf("phrase error: %v", err)) } else { for _, res := range RunChecks(c, pn.Body, pn.Mood) { if res.Pass { rep.ByCheck[res.Name]++ continue } o.Failed = append(o.Failed, res.Name) o.Reasons = append(o.Reasons, res.Name+": "+res.Detail) } } o.Pass = len(o.Failed) == 0 if o.Pass { rep.Passed++ } bump(rep.ByRule, ruleFamily(c.Rule), o.Pass) rep.Outcomes = append(rep.Outcomes, o) } sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] }) rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95) if len(lat) > 0 { rep.Max = lat[len(lat)-1] } return rep, nil } // ruleFamily collapses "routine:зарядка" to "routine" so the per-rule table // stays readable however many routines the operator configures. func ruleFamily(rule string) string { if i := strings.IndexByte(rule, ':'); i > 0 { return rule[:i] } return rule } func bump(m map[string]TagStat, key string, pass bool) { if key == "" { return } s := m[key] s.Total++ if pass { s.Passed++ } m[key] = s } // percentile — nearest-rank on a pre-sorted slice. No interpolation: with ~15 // samples an interpolated p95 invents a latency no call actually took. func percentile(sorted []time.Duration, p float64) time.Duration { if len(sorted) == 0 { return 0 } i := int(p * float64(len(sorted))) if i >= len(sorted) { i = len(sorted) - 1 } return sorted[i] } // String renders the comparison table — composite score, then per-check so a // regression names the property it broke, then latency. func (r Report) String() string { var b strings.Builder fmt.Fprintf(&b, "%s: %d/%d cases pass every check (%.1f%%), %d errors\n", r.Name, r.Passed, r.Total, 100*r.Accuracy(), r.Errors) for _, name := range CheckNames { fmt.Fprintf(&b, " %-10s %d/%d\n", name, r.ByCheck[name], r.Total) } fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max) fmt.Fprintf(&b, " by rule: %s\n", renderStats(r.ByRule)) return b.String() } // Failures — per-case detail, sorted by ID so two runs diff cleanly. func (r Report) Failures() string { var b strings.Builder for _, o := range r.sorted() { if o.Pass { continue } fmt.Fprintf(&b, " %s %q\n %s\n", o.Case.ID, o.Body, strings.Join(o.Reasons, "; ")) } return b.String() } // Messages — every generated message verbatim, pass or fail. This is what a // human reads to judge tone; the score only says which checks fired. func (r Report) Messages() string { var b strings.Builder for _, o := range r.sorted() { mark := "ok " if !o.Pass { mark = "FAIL" } fmt.Fprintf(&b, " %s %-22s [%s] %q\n", mark, o.Case.ID, o.Mood, o.Body) } return b.String() } func (r Report) sorted() []Outcome { out := append([]Outcome(nil), r.Outcomes...) sort.Slice(out, func(i, j int) bool { return out[i].Case.ID < out[j].Case.ID }) return out } func renderStats(m map[string]TagStat) string { keys := make([]string, 0, len(m)) for k := range m { keys = append(keys, k) } sort.Strings(keys) parts := make([]string, 0, len(keys)) for _, k := range keys { parts = append(parts, fmt.Sprintf("%s %d/%d", k, m[k].Passed, m[k].Total)) } return strings.Join(parts, " ") }