Files
Maven/internal/router/llmrouter.go
T
kami bd16ca69e5 Let the LLM router answer "unknown" when it cannot route
Chose an 8th enum value over a confidence number: the model already picks
one enum token, so it costs nothing in the grammar, while a score from a
0.8B model would be uncalibrated noise. A refusal returns "no decision"
with no error, which is the fall-through the caller already uses for a
bad parse, so the classifier and its clarify gate take the turn.

Reviewers: the prompt's counter-examples matter most — a small model will
over-use any easy escape hatch. The training workspace copy of the prompt
still needs the same edit (Vikunja #362).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CGeSZxh1DCtRxmFVSYVGvJ
2026-07-31 11:35:37 +04:00

195 lines
10 KiB
Go

package router
import (
"context"
"encoding/json"
"fmt"
"strings"
"time"
"github.com/kami/maven/internal/llm"
)
// Completer — the LLM seam (mockable). *llm.Client satisfies it.
type Completer interface {
Complete(ctx context.Context, r llm.Req) (string, error)
}
// LLMRouter — the agentic router. One grammar-constrained call classifies the
// utterance and pulls raw slots; deterministic parsers (time) refine downstream.
type LLMRouter struct{ c Completer }
func NewLLMRouter(c Completer) *LLMRouter { return &LLMRouter{c: c} }
// routeGrammar — GBNF constraining the model to a JSON ARRAY of fixed-shape
// action objects (one per ask; compound utterances → multiple). Enum + key set
// prevent free-form drift from a sub-1B model. The string rule is length-bounded
// so a repetition loop cannot fill the whole token budget with one field and
// truncate the JSON.
const routeGrammar = `
root ::= "[" ws action ("," ws action)* ws "]"
action ::= "{" ws "\"intent\"" ws ":" ws intent ("," ws field)* ws "}"
intent ::= "\"fact\"" | "\"reminder\"" | "\"note\"" | "\"query\"" | "\"act\"" | "\"chat\"" | "\"system\"" | "\"unknown\""
field ::= key ws ":" ws string
key ::= "\"key\"" | "\"value\"" | "\"text\"" | "\"verb\""
string ::= "\"" ([^"\\] | "\\" .){0,120} "\""
ws ::= [ \t\n]*
`
// routeSystem — the router prompt. Changed 31-07-2026: the query test now sits
// above the fact test and there is an explicit question test. Before that, a
// question naming a fact key ("сколько воды я выпил с утра") matched the fact
// rule first and was stored as an assertion — 15 of 76 fixture cases.
//
// Changed again 31-07-2026: added the "unknown" escape hatch so the model can
// admit it cannot route (Vikunja #359).
//
// The training workspace keeps its own copy of this prompt for relabelling, and
// `llm/check_prompt_parity.py` there compares the two. That copy is in another
// repo and was not touched, so parity will fail until it gets the same edits —
// both the rule reorder and the "unknown" wording (Vikunja #362).
const routeSystem = `Классифицируй ровно одно сообщение пользователя. Верни ОДИН JSON-массив действий.
Ровно одно намерение: fact, reminder, note, query, act, chat, system.
Есть восьмое значение unknown — только для случаев, когда просьбу невозможно понять.
Классифицируй по цели пользователя. Порядок решения:
1. Хочет напоминание в будущем → reminder
2. Явно просит сохранить информацию → note
3. Задаёт вопрос: есть вопросительное слово (сколько, что, какой, когда, где, кто, почему, как) или знак «?» → query
4. Хочет получить информацию, в том числе о своих же данных → query
5. Утверждает: сообщает или обновляет текущее состояние/событие → fact
6. Просит выполнить работу → act
7. Про ассистента, настройки или память → system
8. Реплика — обрывок или указание на неназванное («это», «то», «потом»), и без него непонятно, что именно нужно сделать → unknown
9. Иначе → chat
Различия:
- note — сохранить информацию, без напоминания. text = суть.
- reminder — уведомить позже. text = что напомнить.
- fact — неявное обновление: пользователь сообщает, что что-то в мире изменилось (текущее/изменённое состояние, случившееся событие). key/value.
- unknown — редкий случай. Ставь его, только если в самой реплике нет ни предмета, ни действия. Короткая, простая или незнакомая тема — это не причина для unknown: приветствие и болтовня — это chat, вопрос на любую тему — это query, просьба сделать что-то названное — это act.
- query против fact — решает форма реплики, а не тема. Вопрос о состоянии — это query, даже если названо то же самое, что бывает в fact. Только утверждение — это fact.
Примеры:
"запиши пароль" → {"intent":"note","text":"пароль"}
"напомни купить молоко" → {"intent":"reminder","text":"купить молоко"}
"запиши купить молоко" → {"intent":"note","text":"купить молоко"}
"я выпил воду" → {"intent":"fact","key":"water","value":"выпил"}
"сколько воды я выпил с утра" → {"intent":"query","text":"сколько воды я выпил с утра"}
"сколько раз я ел вчера?" → {"intent":"query","text":"сколько раз я ел вчера"}
"мой любимый фильм — Интерстеллар" → {"intent":"note","text":"любимый фильм — Интерстеллар"}
"что такое docker?" → {"intent":"query","text":"что такое docker"}
"напиши письмо" → {"intent":"act","verb":"написать письмо"}
"очисти память" → {"intent":"system"}
"привет" → {"intent":"chat","text":"привет"}
"сделай это" → {"intent":"unknown"}
"ну это" → {"intent":"unknown"}
"потом" → {"intent":"unknown"}
Но не путай — здесь unknown не нужен:
"сделай кофе" → {"intent":"act","verb":"сделать кофе"}
"что такое кватернион?" → {"intent":"query","text":"что такое кватернион"}
"ага" → {"intent":"chat","text":"ага"}
Ответ — JSON-массив: по одному объекту на каждую просьбу. Обычно один. Если в реплике несколько просьб — по объекту на каждую. "напомни купить молоко, и запиши что кофе кончился" → [{"intent":"reminder","text":"купить молоко"},{"intent":"note","text":"кофе кончился"}]. Только JSON, без пояснений.`
// routeRepeatPenalty — the sub-1B model loops one sentence inside the text field
// until it runs out of tokens, which truncates the JSON. 1.15 is enough to break
// the loop without hurting short slot values.
const routeRepeatPenalty = 1.15
// routeIntentUnknown — the model's way of saying "I could not route this".
// It is a wire value only: it never becomes a router.Intent, it just makes
// Route return ok=false so the caller drops to the classifier cascade.
const routeIntentUnknown = "unknown"
type routeAction struct {
Intent string `json:"intent"`
Key string `json:"key"`
Value string `json:"value"`
Text string `json:"text"`
Verb string `json:"verb"`
}
// Route asks the model for one decision. The bool is false when there is no
// decision to use: either the model failed (err set) or it refused with the
// "unknown" intent (err nil). Both mean the same thing to the caller — use the
// classifier instead.
func (lr *LLMRouter) Route(ctx context.Context, utterance string, now time.Time) (Decision, bool, error) {
raw, err := lr.c.Complete(ctx, llm.Req{
System: routeSystem,
User: utterance,
Grammar: routeGrammar,
MaxTokens: 128,
RepeatPenalty: routeRepeatPenalty,
})
if err != nil {
return Decision{}, false, err
}
acts, err := parseActions(strings.TrimSpace(raw))
if err != nil {
return Decision{}, false, fmt.Errorf("llmrouter: parse %q: %w", raw, err)
}
if len(acts) == 0 {
return Decision{}, false, fmt.Errorf("llmrouter: empty action list %q", raw)
}
// ponytail: contract is an array (compound utterances → N actions), but the
// Router cascade still returns one Decision. Dispatch of all actions lands
// with the engine turn-on (Router.Route → []Decision, both voice.go handlers
// loop). Until then only the first ask is honored.
a := acts[0]
// The model refused. Report "no decision" without an error, which is the
// same fall-through the caller already uses for a parse failure — the
// classifier cascade gets the turn and its own confidence gate decides
// whether to ask. Better a slower second opinion than a confident guess.
if a.Intent == routeIntentUnknown {
return Decision{}, false, nil
}
d := Decision{Utterance: utterance, Stage: 1, Confidence: 1.0}
switch Intent(a.Intent) {
case IntentFact:
d.Intent = IntentFact
d.Slots.Key, d.Slots.Value = a.Key, a.Value
d.Slots.HasKey = a.Key != ""
case IntentReminder:
d.Intent = IntentReminder
d.Slots.Text = firstNonEmpty(a.Text, utterance)
case IntentNote:
d.Intent = IntentNote
d.Slots.Text = firstNonEmpty(a.Text, utterance)
case IntentQuery:
d.Intent = IntentQuery
d.Slots.Text = firstNonEmpty(a.Text, utterance)
case IntentAct:
d.Intent = IntentAct
d.Slots.Text = firstNonEmpty(a.Verb, utterance)
case IntentSystem:
d.Intent = IntentSystem
default:
d.Intent = IntentChat
d.Slots.Text = firstNonEmpty(a.Text, utterance)
}
return d, true, nil
}
// parseActions accepts the array contract (`[{...},...]`) or a bare object
// (`{...}`) for robustness against a model that drops the wrapper.
func parseActions(raw string) ([]routeAction, error) {
if strings.HasPrefix(raw, "[") {
var as []routeAction
return as, json.Unmarshal([]byte(raw), &as)
}
var a routeAction
if err := json.Unmarshal([]byte(raw), &a); err != nil {
return nil, err
}
return []routeAction{a}, nil
}
func firstNonEmpty(a, b string) string {
if strings.TrimSpace(a) != "" {
return a
}
return b
}