6d3f5b5b01
Slots.Text was the raw utterance for every intent, so a reminder could not have an empty subject. StillMissing never reported SlotText, the question "О чём напомнить?" was unaskable, and the branch in PendingQuestion.Answer that fills a text slot could only overwrite the whole request. The LLM path now keeps the model's own text, empty included, and the gate turns a subjectless reminder into a question. The classifier path is unchanged: it has no subject parser, so the utterance is the only signal it has.
206 lines
8.4 KiB
Go
206 lines
8.4 KiB
Go
package router
|
||
|
||
import (
|
||
"context"
|
||
"log"
|
||
"time"
|
||
)
|
||
|
||
// Config — wires the cascade. Build via New; a zero-value Router is unusable.
|
||
type Config struct {
|
||
// Grammars — stage-0 exact-match rules. DefaultGrammars(actMatcher) wires
|
||
// the wake-word act fast path; the daemon may append more.
|
||
Grammars []Grammar
|
||
// Classifier — stage-1 nearest-centroid classifier. Must be seeded with
|
||
// ~10 examples/intent at bootstrap (per spec) before free-form routing
|
||
// is trustworthy; until then Route returns ErrNoIntents on free-form input.
|
||
Classifier *Classifier
|
||
// Extractor — stage-2 per-intent slot extraction. Any nil sub-parser just
|
||
// leaves the corresponding Has* flag false for that intent.
|
||
Extractor Extractor
|
||
// Threshold — stage-3 confidence gate. Below ⇒ Clarify, don't guess. The
|
||
// spec leaves this open (defines how often maven asks vs guesses on free-
|
||
// form input; the whole reactive mvp feel rides on it). The daemon sets it.
|
||
Threshold float64
|
||
// LLM — optional agentic router. When set, Route consults it after stage-0
|
||
// and before the classifier cascade, classifying the utterance via a
|
||
// grammar-constrained call to the resident model (Qwen3-1.7B). On any
|
||
// error/parse failure, falls through to the classifier (never fails the
|
||
// turn on the model).
|
||
LLM *LLMRouter
|
||
}
|
||
|
||
// Router — the deterministic cascade. Route never guesses: stage 0 wins
|
||
// outright, stage 1 scores, stage 2 extracts, stage 3 gates. The SLM only
|
||
// phrases the reply — it never owns the route.
|
||
type Router struct {
|
||
grammars []Grammar
|
||
classifier *Classifier
|
||
extractor Extractor
|
||
threshold float64
|
||
llm *LLMRouter
|
||
}
|
||
|
||
func New(cfg Config) *Router {
|
||
return &Router{
|
||
grammars: cfg.Grammars,
|
||
classifier: cfg.Classifier,
|
||
extractor: cfg.Extractor,
|
||
threshold: cfg.Threshold,
|
||
llm: cfg.LLM,
|
||
}
|
||
}
|
||
|
||
// Route — the cascade: stage 0 (exact match) → 1 (classify) → 2 (extract) →
|
||
// 3 (confidence gate).
|
||
//
|
||
// Stage 0 wins outright: returns at confidence 1.0, no classifier.
|
||
// Otherwise the classifier scores every intent; the best wins; slots are
|
||
// extracted for that intent. If the winning score < threshold the Decision
|
||
// is flagged Clarify (the daemon asks rather than guesses — same shape as
|
||
// since(key)==null → don't fire: a misrouted fact is a confident wrong write,
|
||
// worse than a gap).
|
||
func (r *Router) Route(ctx context.Context, utterance string, now time.Time) (Decision, error) {
|
||
// stage 0 — exact match / grammar. First match wins; grammars are ordered.
|
||
// Grammars like time/date/reminder don't expect a wake-word prefix, but
|
||
// the STT often includes one (transcribed phonetically, any script) — try
|
||
// the wake-stripped utterance too so those grammars still fire.
|
||
stripped, hadWake := StripWakeToken(utterance)
|
||
for _, g := range r.grammars {
|
||
m := g.Pattern.FindStringSubmatch(utterance)
|
||
if m == nil && hadWake {
|
||
m = g.Pattern.FindStringSubmatch(stripped)
|
||
}
|
||
if m == nil {
|
||
continue
|
||
}
|
||
d, ok := g.Build(m)
|
||
if !ok {
|
||
continue // grammar matched shape but not content → fall through
|
||
}
|
||
d.Utterance = utterance
|
||
return d, nil
|
||
}
|
||
|
||
// stage 1a — LLM router (when wired). It reasons over the utterance instead
|
||
// of nearest-centroid guessing. On any error/parse-fail, fall through to the
|
||
// classifier cascade (never fail the turn on the model).
|
||
if r.llm != nil {
|
||
if d, ok, err := r.llm.Route(ctx, utterance, now); err == nil && ok {
|
||
d.Utterance = utterance
|
||
r.fillSlots(ctx, &d, now)
|
||
r.gateLLMDecision(&d)
|
||
return d, nil
|
||
} else if err != nil {
|
||
log.Printf("router: llm route fell back to classifier: %v", err)
|
||
}
|
||
}
|
||
|
||
// stage 1 — intent classifier.
|
||
results, err := r.classifier.Classify(ctx, utterance)
|
||
if err != nil {
|
||
return Decision{}, err
|
||
}
|
||
best := results[0]
|
||
|
||
// stage 2 — slot extraction for the winning intent.
|
||
d := Decision{
|
||
Utterance: utterance,
|
||
Stage: 2,
|
||
Intent: best.Intent,
|
||
Confidence: best.Score,
|
||
Slots: r.extractor.Extract(ctx, best.Intent, utterance, now),
|
||
}
|
||
|
||
// stage 3 — confidence gate. Below threshold ⇒ clarify, don't guess.
|
||
if d.Confidence < r.threshold {
|
||
d.Stage = 3
|
||
d.Clarify = true
|
||
}
|
||
return d, nil
|
||
}
|
||
|
||
// fillSlots — run stage-2 extraction on an LLM decision and fill only the slots
|
||
// the model left empty. The LLM wins where it answered: it saw the sentence, the
|
||
// parsers are keyword tables. Extraction covers what the model cannot produce at
|
||
// all — a parsed reminder time and an allowlist fn.
|
||
//
|
||
// If a reminder still has no time, leave it missing. The daemon then says it
|
||
// could not read the time; inventing one would set a wrong alarm.
|
||
func (r *Router) fillSlots(ctx context.Context, d *Decision, now time.Time) {
|
||
ex := r.extractor.Extract(ctx, d.Intent, d.Utterance, now)
|
||
if !d.Slots.HasTime && ex.HasTime {
|
||
d.Slots.Time, d.Slots.HasTime = ex.Time, ex.HasTime
|
||
}
|
||
if !d.Slots.HasKey && ex.HasKey {
|
||
d.Slots.Key, d.Slots.Value, d.Slots.HasKey = ex.Key, ex.Value, ex.HasKey
|
||
}
|
||
if !d.Slots.HasFn && ex.HasFn {
|
||
d.Slots.Fn, d.Slots.Args, d.Slots.HasFn = ex.Fn, ex.Args, ex.HasFn
|
||
}
|
||
// For an act the model returns the verb in Text ("restart nginx"), which is
|
||
// often cleaner than the raw utterance ("maven, could you restart nginx").
|
||
// Try it too when the utterance did not match the allowlist.
|
||
if d.Intent == IntentAct && !d.Slots.HasFn && r.extractor.Acts != nil &&
|
||
d.Slots.Text != "" && d.Slots.Text != d.Utterance {
|
||
if fn, args, ok := r.extractor.Acts.Match(d.Slots.Text); ok {
|
||
d.Slots.Fn, d.Slots.Args, d.Slots.HasFn = fn, args, true
|
||
}
|
||
}
|
||
// The extractor's Text is the raw utterance, which is the payload for a
|
||
// note, a query or a chat turn but not for a reminder — there Text is the
|
||
// subject, what she says at the hour. Backfilling it made Text impossible
|
||
// to be empty, so StillMissing never reported SlotText and "О чём
|
||
// напомнить?" was unaskable; the answer to a question she did manage to
|
||
// ask then overwrote the whole request instead of filling one gap
|
||
// (Vikunja #383). A reminder with no subject stays empty and is gated
|
||
// below into a question.
|
||
if d.Slots.Text == "" && d.Intent != IntentReminder {
|
||
d.Slots.Text = ex.Text
|
||
}
|
||
// Stage stays 1: it says who decided the route, and that was the LLM.
|
||
}
|
||
|
||
// gateLLMDecision — stage 3 for the LLM path (Vikunja #359). This used to be
|
||
// the classifier's job alone (see the threshold check at the bottom of
|
||
// Route): the LLM branch returned straight from fillSlots and never touched
|
||
// r.threshold at all, so a hardcoded Confidence: 1.0 in llmrouter.go could
|
||
// never gate. Two more structural holes are checked here, after fillSlots
|
||
// has had a chance to fill them from the deterministic parsers — checking
|
||
// before fillSlots would flag e.g. every keyless fact the fact parser goes
|
||
// on to resolve (TestLLMFactGetsKeyFromParser):
|
||
// - a fact with no key even after the parser tried — nothing to write, or
|
||
// worse, a confident write under the wrong key;
|
||
// - an act that never resolved to an allowlisted fn — a confident guess
|
||
// here means either silently doing nothing or, if the daemon is lax,
|
||
// running something never on the allowlist. Don't guess; ask.
|
||
//
|
||
// Anything below threshold gets the exact same Clarify=true treatment the
|
||
// classifier path already produces — same field, same daemon-side consumer
|
||
// (cmd/mavend/clarify.go), nothing new to wire.
|
||
func (r *Router) gateLLMDecision(d *Decision) {
|
||
if d.Intent == IntentFact && !d.Slots.HasKey && d.Confidence > llmThinConfidence {
|
||
d.Confidence = llmThinConfidence
|
||
}
|
||
if d.Intent == IntentAct && !d.Slots.HasFn && d.Confidence > llmThinConfidence {
|
||
d.Confidence = llmThinConfidence
|
||
}
|
||
// A reminder with no subject: she knows when but not what to say then.
|
||
// Setting it anyway fires an empty reminder at the hour, which reads as a
|
||
// bug to him and cannot be repaired after the fact. Ask (Vikunja #383).
|
||
if d.Intent == IntentReminder && d.Slots.Text == "" && d.Confidence > llmThinConfidence {
|
||
d.Confidence = llmThinConfidence
|
||
}
|
||
if d.Confidence < r.threshold {
|
||
d.Clarify = true
|
||
}
|
||
}
|
||
|
||
// CorrectMisroute — the user corrected a bad classification. Appends a new
|
||
// example for the corrected intent (append-only — grows the classifier, no
|
||
// retrain). Same shape as nudges.outcome tuning cooldowns: more reliable over
|
||
// time, introspectable, no model surgery.
|
||
func (r *Router) CorrectMisroute(ctx context.Context, utterance string, corrected Intent) error {
|
||
return r.classifier.AddExample(ctx, corrected, utterance)
|
||
}
|