f0f7ebc9b2
Confidence was hardcoded to 1.0 for every LLM decision, and the LLM branch
in Router.Route returned straight from fillSlots without ever touching the
stage-3 threshold gate — so the LLM path could not produce a Clarify no
matter what confidence a model reported. That is why all 6 want_clarify
cases in the 77-case RU fixture were missed by every model in the bake-off.
Fix reads structural signal instead of changing the (parity-locked) router
prompt: a single-token utterance ("вода", "бэкап") is flagged thin evidence
in llmrouter.go; a fact left keyless or an act that never resolves to an
allowlisted fn, checked after fillSlots so the deterministic parsers get
first crack, is flagged in router.go's new gateLLMDecision. Anything below
config.DefaultRouterThreshold (0.55) now sets Clarify=true through the same
path the classifier already uses.
Added unit tests with a stubbed Completer proving both directions: thin
cases clarify, clean multi-word/resolved-slot cases stay confident. The
77-case fixture re-run against a live llama-server is still needed to
confirm the 6/6 moves — not done here, no llama-server on this box.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
251 lines
14 KiB
Go
251 lines
14 KiB
Go
package router
|
||
|
||
import (
|
||
"context"
|
||
"encoding/json"
|
||
"fmt"
|
||
"strings"
|
||
"time"
|
||
|
||
"github.com/kami/maven/internal/llm"
|
||
)
|
||
|
||
// Completer — the LLM seam (mockable). *llm.Client satisfies it.
|
||
type Completer interface {
|
||
Complete(ctx context.Context, r llm.Req) (string, error)
|
||
}
|
||
|
||
// LLMRouter — the agentic router. One grammar-constrained call classifies the
|
||
// utterance and pulls raw slots; deterministic parsers (time) refine downstream.
|
||
type LLMRouter struct{ c Completer }
|
||
|
||
func NewLLMRouter(c Completer) *LLMRouter { return &LLMRouter{c: c} }
|
||
|
||
// routeGrammar — GBNF constraining the model to a JSON ARRAY of fixed-shape
|
||
// action objects (one per ask; compound utterances → multiple). Enum + key set
|
||
// prevent free-form drift from a sub-1B model. The string rule is length-bounded
|
||
// so a repetition loop cannot fill the whole token budget with one field and
|
||
// truncate the JSON.
|
||
const routeGrammar = `
|
||
root ::= "[" ws action ("," ws action)* ws "]"
|
||
action ::= "{" ws "\"intent\"" ws ":" ws intent ("," ws field)* ws "}"
|
||
intent ::= "\"fact\"" | "\"reminder\"" | "\"note\"" | "\"query\"" | "\"act\"" | "\"chat\"" | "\"system\"" | "\"unknown\""
|
||
field ::= key ws ":" ws string
|
||
key ::= "\"key\"" | "\"value\"" | "\"text\"" | "\"verb\""
|
||
string ::= "\"" ([^"\\] | "\\" .){0,120} "\""
|
||
ws ::= [ \t\n]*
|
||
`
|
||
|
||
// routeSystem — the router prompt. Changed 31-07-2026: the query test now sits
|
||
// above the fact test and there is an explicit question test. Before that, a
|
||
// question naming a fact key ("сколько воды я выпил с утра") matched the fact
|
||
// rule first and was stored as an assertion — 15 of 76 fixture cases.
|
||
//
|
||
// Changed again 31-07-2026: added the "unknown" escape hatch so the model can
|
||
// admit it cannot route (Vikunja #359).
|
||
//
|
||
// Changed again 31-07-2026: added the clock/calendar rule (Vikunja #374). The
|
||
// prompt never said which side "который час" or "какое число завтра" belong on,
|
||
// so the model guessed — `system→query ×4` in every eval run. The rule sits
|
||
// above the question test on purpose: these utterances all carry a question
|
||
// word, so a later rule would never be reached. The boundary is what the
|
||
// daemon can actually answer: only replySystem in cmd/mavend/voice.go owns the
|
||
// clock and the calendar formatter, while the agenda ("что у меня завтра") is
|
||
// answered inside the query branch, so that side stays query.
|
||
//
|
||
// The training workspace keeps its own copy of this prompt for relabelling, and
|
||
// `llm/check_prompt_parity.py` there compares the two. That copy is in another
|
||
// repo and was not touched, so parity will fail until it gets the same edits —
|
||
// both the rule reorder and the "unknown" wording (Vikunja #362) — and now the
|
||
// clock/calendar rule too. The training workspace is not checked out on this
|
||
// box at all, so it could not be updated here; #362 still covers the catch-up.
|
||
const routeSystem = `Классифицируй ровно одно сообщение пользователя. Верни ОДИН JSON-массив действий.
|
||
|
||
Ровно одно намерение: fact, reminder, note, query, act, chat, system.
|
||
Есть восьмое значение unknown — только для случаев, когда просьбу невозможно понять.
|
||
|
||
Классифицируй по цели пользователя. Порядок решения:
|
||
1. Хочет напоминание в будущем → reminder
|
||
2. Явно просит сохранить информацию → note
|
||
3. Спрашивает только «который час» / «какое число» / «какой день недели» — сами часы или календарная дата, без своих данных → system
|
||
4. Задаёт вопрос: есть вопросительное слово (сколько, что, какой, когда, где, кто, почему, как) или знак «?» → query
|
||
5. Хочет получить информацию, в том числе о своих же данных → query
|
||
6. Утверждает: сообщает или обновляет текущее состояние/событие → fact
|
||
7. Просит выполнить работу → act
|
||
8. Про ассистента, настройки или память → system
|
||
9. Реплика — обрывок или указание на неназванное («это», «то», «потом»), и без него непонятно, что именно нужно сделать → unknown
|
||
10. Иначе → chat
|
||
|
||
Различия:
|
||
- note — сохранить информацию, без напоминания. text = суть.
|
||
- reminder — уведомить позже. text = что напомнить.
|
||
- fact — неявное обновление: пользователь сообщает, что что-то в мире изменилось (текущее/изменённое состояние, случившееся событие). key/value.
|
||
- unknown — редкий случай. Ставь его, только если в самой реплике нет ни предмета, ни действия. Короткая, простая или незнакомая тема — это не причина для unknown: приветствие и болтовня — это chat, вопрос на любую тему — это query, просьба сделать что-то названное — это act.
|
||
- system против query — часы и календарная дата сами по себе (сколько времени, какое число, какой день недели — можно и про завтра, и про другой город) — это system. А что записано в календаре или в памяти («что у меня завтра», «какие есть напоминания») — это query. Если в реплике есть просьба (напомни, запиши, сделай), то названное время — просто деталь просьбы, и это не system.
|
||
- query против fact — решает форма реплики, а не тема. Вопрос о состоянии — это query, даже если названо то же самое, что бывает в fact. Только утверждение — это fact.
|
||
|
||
Примеры:
|
||
"запиши пароль" → {"intent":"note","text":"пароль"}
|
||
"напомни купить молоко" → {"intent":"reminder","text":"купить молоко"}
|
||
"запиши купить молоко" → {"intent":"note","text":"купить молоко"}
|
||
"я выпил воду" → {"intent":"fact","key":"water","value":"выпил"}
|
||
"сколько воды я выпил с утра" → {"intent":"query","text":"сколько воды я выпил с утра"}
|
||
"сколько раз я ел вчера?" → {"intent":"query","text":"сколько раз я ел вчера"}
|
||
"мой любимый фильм — Интерстеллар" → {"intent":"note","text":"любимый фильм — Интерстеллар"}
|
||
"что такое docker?" → {"intent":"query","text":"что такое docker"}
|
||
"напиши письмо" → {"intent":"act","verb":"написать письмо"}
|
||
"очисти память" → {"intent":"system"}
|
||
"который час?" → {"intent":"system"}
|
||
"какое число завтра?" → {"intent":"system"}
|
||
"привет" → {"intent":"chat","text":"привет"}
|
||
"сделай это" → {"intent":"unknown"}
|
||
"ну это" → {"intent":"unknown"}
|
||
"потом" → {"intent":"unknown"}
|
||
Но не путай — здесь unknown не нужен:
|
||
"сделай кофе" → {"intent":"act","verb":"сделать кофе"}
|
||
"что такое кватернион?" → {"intent":"query","text":"что такое кватернион"}
|
||
"ага" → {"intent":"chat","text":"ага"}
|
||
|
||
Ответ — JSON-массив: по одному объекту на каждую просьбу. Обычно один. Если в реплике несколько просьб — по объекту на каждую. "напомни купить молоко, и запиши что кофе кончился" → [{"intent":"reminder","text":"купить молоко"},{"intent":"note","text":"кофе кончился"}]. Только JSON, без пояснений.`
|
||
|
||
// routeRepeatPenalty — the sub-1B model loops one sentence inside the text field
|
||
// until it runs out of tokens, which truncates the JSON. 1.15 is enough to break
|
||
// the loop without hurting short slot values.
|
||
const routeRepeatPenalty = 1.15
|
||
|
||
// routeIntentUnknown — the model's way of saying "I could not route this".
|
||
// It is a wire value only: it never becomes a router.Intent, it just makes
|
||
// Route return ok=false so the caller drops to the classifier cascade.
|
||
const routeIntentUnknown = "unknown"
|
||
|
||
// llmFullConfidence / llmThinConfidence — Vikunja #359. Confidence used to be
|
||
// hardcoded to 1.0 for every LLM decision, so the stage-3 gate in router.go
|
||
// never had anything to bite on and the LLM path could never produce a
|
||
// Clarify: on the 77-case RU fixture, 6/6 want_clarify cases were missed by
|
||
// EVERY model in the 31-07-2026 bake-off (0.8B through 2B) — proof this was a
|
||
// code bug, not a capability ceiling.
|
||
//
|
||
// The fix does not touch the prompt (routeSystem is under
|
||
// llm/check_prompt_parity.py in the training workspace; changing its text
|
||
// creates a parity break that has to be fixed there too — see Vikunja #362).
|
||
// Instead it reads structural signal that is already free:
|
||
// - a single-token utterance is thin evidence for anything a grammar
|
||
// didn't already catch at stage 0 — "вода" and "бэкап" alone don't say
|
||
// fact-vs-query or act-vs-report;
|
||
// - a fact with no key, or an act that never resolves to an allowlisted fn
|
||
// (checked in router.go, after slot-fill has had its say), is a decision
|
||
// with a hole in the one slot that makes it actionable.
|
||
//
|
||
// A model self-reporting confidence in the JSON was considered and rejected:
|
||
// a sub-2B is not calibrated (nothing stops it saying "confident" on exactly
|
||
// the cases it gets wrong today), and true logprobs would need a response
|
||
// field internal/llm.Client's Complete does not currently return — see
|
||
// internal/llm/client.go.
|
||
//
|
||
// llmThinConfidence sits below config.DefaultRouterThreshold (0.55) so the
|
||
// existing stage-3 gate in Router.Route treats it exactly like a low-scoring
|
||
// classifier result — same lane, same daemon-side clarify machinery
|
||
// (cmd/mavend/clarify.go), no new consumer to build.
|
||
const (
|
||
llmFullConfidence = 1.0
|
||
llmThinConfidence = 0.3
|
||
)
|
||
|
||
type routeAction struct {
|
||
Intent string `json:"intent"`
|
||
Key string `json:"key"`
|
||
Value string `json:"value"`
|
||
Text string `json:"text"`
|
||
Verb string `json:"verb"`
|
||
}
|
||
|
||
// Route asks the model for one decision. The bool is false when there is no
|
||
// decision to use: either the model failed (err set) or it refused with the
|
||
// "unknown" intent (err nil). Both mean the same thing to the caller — use the
|
||
// classifier instead.
|
||
func (lr *LLMRouter) Route(ctx context.Context, utterance string, now time.Time) (Decision, bool, error) {
|
||
raw, err := lr.c.Complete(ctx, llm.Req{
|
||
System: routeSystem,
|
||
User: utterance,
|
||
Grammar: routeGrammar,
|
||
MaxTokens: 128,
|
||
RepeatPenalty: routeRepeatPenalty,
|
||
})
|
||
if err != nil {
|
||
return Decision{}, false, err
|
||
}
|
||
acts, err := parseActions(strings.TrimSpace(raw))
|
||
if err != nil {
|
||
return Decision{}, false, fmt.Errorf("llmrouter: parse %q: %w", raw, err)
|
||
}
|
||
if len(acts) == 0 {
|
||
return Decision{}, false, fmt.Errorf("llmrouter: empty action list %q", raw)
|
||
}
|
||
// ponytail: contract is an array (compound utterances → N actions), but the
|
||
// Router cascade still returns one Decision. Dispatch of all actions lands
|
||
// with the engine turn-on (Router.Route → []Decision, both voice.go handlers
|
||
// loop). Until then only the first ask is honored.
|
||
a := acts[0]
|
||
// The model refused. Report "no decision" without an error, which is the
|
||
// same fall-through the caller already uses for a parse failure — the
|
||
// classifier cascade gets the turn and its own confidence gate decides
|
||
// whether to ask. Better a slower second opinion than a confident guess.
|
||
if a.Intent == routeIntentUnknown {
|
||
return Decision{}, false, nil
|
||
}
|
||
d := Decision{Utterance: utterance, Stage: 1, Confidence: llmFullConfidence}
|
||
// A single-token utterance is thin evidence: the model had nothing to
|
||
// disambiguate on ("вода" is a fact-or-query coin flip, "бэкап" an
|
||
// act-or-report one) and stage 0 would already have won on anything
|
||
// that pattern-matches cleanly. Flag it now; router.go's stage-3 gate
|
||
// (Router.Route) decides whether that trips Clarify.
|
||
if len(strings.Fields(utterance)) <= 1 {
|
||
d.Confidence = llmThinConfidence
|
||
}
|
||
switch Intent(a.Intent) {
|
||
case IntentFact:
|
||
d.Intent = IntentFact
|
||
d.Slots.Key, d.Slots.Value = a.Key, a.Value
|
||
d.Slots.HasKey = a.Key != ""
|
||
case IntentReminder:
|
||
d.Intent = IntentReminder
|
||
d.Slots.Text = firstNonEmpty(a.Text, utterance)
|
||
case IntentNote:
|
||
d.Intent = IntentNote
|
||
d.Slots.Text = firstNonEmpty(a.Text, utterance)
|
||
case IntentQuery:
|
||
d.Intent = IntentQuery
|
||
d.Slots.Text = firstNonEmpty(a.Text, utterance)
|
||
case IntentAct:
|
||
d.Intent = IntentAct
|
||
d.Slots.Text = firstNonEmpty(a.Verb, utterance)
|
||
case IntentSystem:
|
||
d.Intent = IntentSystem
|
||
default:
|
||
d.Intent = IntentChat
|
||
d.Slots.Text = firstNonEmpty(a.Text, utterance)
|
||
}
|
||
return d, true, nil
|
||
}
|
||
|
||
// parseActions accepts the array contract (`[{...},...]`) or a bare object
|
||
// (`{...}`) for robustness against a model that drops the wrapper.
|
||
func parseActions(raw string) ([]routeAction, error) {
|
||
if strings.HasPrefix(raw, "[") {
|
||
var as []routeAction
|
||
return as, json.Unmarshal([]byte(raw), &as)
|
||
}
|
||
var a routeAction
|
||
if err := json.Unmarshal([]byte(raw), &a); err != nil {
|
||
return nil, err
|
||
}
|
||
return []routeAction{a}, nil
|
||
}
|
||
|
||
func firstNonEmpty(a, b string) string {
|
||
if strings.TrimSpace(a) != "" {
|
||
return a
|
||
}
|
||
return b
|
||
}
|