package router import ( "context" "errors" "log" "time" "github.com/kami/maven/internal/decision" ) // Config — wires the cascade. Build via New; a zero-value Router is unusable. type Config struct { // Grammars — stage-0 exact-match rules. DefaultGrammars(actMatcher) wires // the wake-word act fast path; the daemon may append more. Grammars []Grammar // Classifier — stage-1 nearest-centroid classifier. Must be seeded with // ~10 examples/intent at bootstrap (per spec) before free-form routing // is trustworthy; until then Route returns ErrNoIntents on free-form input. Classifier *Classifier // Extractor — stage-2 per-intent slot extraction. Any nil sub-parser just // leaves the corresponding Has* flag false for that intent. Extractor Extractor // Threshold — stage-3 confidence gate. Below ⇒ Clarify, don't guess. The // spec leaves this open (defines how often maven asks vs guesses on free- // form input; the whole reactive mvp feel rides on it). The daemon sets it. Threshold float64 // LLM — optional agentic router. When set, Route consults it after stage-0 // and before the classifier cascade, classifying the utterance via a // grammar-constrained call to the resident model (Qwen3-1.7B). On any // error/parse failure, falls through to the classifier (never fails the // turn on the model). LLM *LLMRouter // Heads — optional routing heads over the fine-tuned embedder copy. When // set, Route consults them after stage 0 and before the LLM router. They // decline below their own confidence threshold, so a low-confidence turn // reaches the model exactly as it does today. Nil is the shipped-before // behaviour and costs nothing. Heads *RouterHeads } // Router — the deterministic cascade. Route never guesses: stage 0 wins // outright, stage 1 scores, stage 2 extracts, stage 3 gates. The SLM only // phrases the reply — it never owns the route. type Router struct { grammars []Grammar classifier *Classifier extractor Extractor threshold float64 llm *LLMRouter heads *RouterHeads } func New(cfg Config) *Router { return &Router{ grammars: cfg.Grammars, classifier: cfg.Classifier, extractor: cfg.Extractor, threshold: cfg.Threshold, llm: cfg.LLM, heads: cfg.Heads, } } // Route — the cascade: stage 0 (exact match) → 1 (classify) → 2 (extract) → // 3 (confidence gate). // // Stage 0 wins outright: returns at confidence 1.0, no classifier. // Otherwise the classifier scores every intent; the best wins; slots are // extracted for that intent. If the winning score < threshold the Decision // is flagged Clarify (the daemon asks rather than guesses — same shape as // since(key)==null → don't fire: a misrouted fact is a confident wrong write, // worse than a gap). func (r *Router) Route(ctx context.Context, input NormalizedInput, now time.Time) (Decision, error) { utterance := input.Text // stage 0 — deterministic fast path. First match wins; grammars are ordered. fast, err := r.TryFastPath(ctx, input, now) if err != nil { return Decision{}, err } if fast.Matched { return fast.Decision, nil } // stage 0b — routing heads (when wired). A softmax over the label set, so // it cannot name an intent or a destination that does not exist, and its // max is a real confidence. It runs before the model because it is three // orders of magnitude faster and scores better on both halves of the route. // // It declines below its threshold rather than clarifying. A declined turn // carries on to the model and then the classifier, which is what a box with // no weights file does on every turn. if r.heads != nil { res, ok, err := r.heads.Route(ctx, utterance) switch { case err != nil: log.Printf("router: heads fell through to the rest of the cascade: %v", err) decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantHeads, Outcome: decision.Declined, Reason: "error: " + err.Error(), }) case !ok: decision.Note(ctx, decision.Scored(decision.StageRoute, claimantHeads, string(res.Intent), res.Confidence, decision.Declined, "below the heads confidence threshold")) default: d := Decision{ Utterance: utterance, Stage: 2, Intent: res.Intent, Confidence: res.Confidence, Source: res.Source, Clarify: res.Clarify, Producer: RouteProducerHeads, } r.fillSlots(ctx, &d, now) applyCapabilityToSlots(&d, SelectCapability(d, r.extractor.Acts)) // The clarify head relearned the English assumption that one word // cannot be a sentence. Russian verbs carry subject and tense, and a // deterministic fact parser which also found a key gives both halves // of a complete write. That structural evidence outranks this one // learned veto; question-shaped turns remain untouched. clarifyOverruled := d.Clarify && completeParsedSingleVerbFact(d) if clarifyOverruled { d.Clarify = false } decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantLLM, Outcome: decision.NeverAsked, Reason: "the routing heads answered", }) decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantClassifier, Outcome: decision.NeverAsked, Reason: "the routing heads answered", }) outcome, reason := decision.Won, "" if d.Clarify { outcome, reason = decision.Thinned, "the clarify head says there is too little here to act on" } else if clarifyOverruled { reason = "a parsed single-token Russian verb is a complete fact" } decision.Note(ctx, decision.Scored(decision.StageRoute, claimantHeads, string(d.Intent), d.Confidence, outcome, reason)) return d, nil } } else { decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantHeads, Outcome: decision.NeverAsked, Reason: "no routing heads are wired", }) } // stage 1a — LLM router (when wired). It reasons over the utterance instead // of nearest-centroid guessing. On any error/parse-fail, fall through to the // classifier cascade (never fail the turn on the model). if r.llm != nil { if d, ok, err := r.llm.Route(ctx, utterance, now); err == nil && ok { d.Utterance = utterance d.Producer = RouteProducerLLM r.fillSlots(ctx, &d, now) // Capability selection: after fillSlots populates Text, SelectCapability // tries the LLM-cleaned text against the allowlist when the raw // utterance did not match. This replaces the former LLM text backfill // that was embedded in fillSlots. Must run before gateLLMDecision so // the gate sees the correct resolution state for acts. sel := SelectCapability(d, r.extractor.Acts) applyCapabilityToSlots(&d, sel) before := d.Confidence r.gateLLMDecision(&d) // The classifier is the floor and it never ran, which is the whole // reason a wrong LLM route reads as unexplainable (V-564). decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantClassifier, Outcome: decision.NeverAsked, Reason: "the LLM router answered", }) outcome, reason := decision.Won, "" if d.Confidence < before { outcome, reason = decision.Thinned, thinReason(&d) } decision.Note(ctx, decision.Scored(decision.StageRoute, claimantLLM, string(d.Intent), d.Confidence, outcome, reason)) return d, nil } else if err != nil { log.Printf("router: llm route fell back to classifier: %v", err) decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantLLM, Outcome: decision.Declined, Reason: "error: " + err.Error(), }) } else { decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantLLM, Outcome: decision.Declined, Reason: "no parsable route in the reply", }) } } else { decision.Note(ctx, decision.Claim{ Stage: decision.StageRoute, Claimant: claimantLLM, Outcome: decision.NeverAsked, Reason: "no LLM router is wired", }) } // stage 1 — intent classifier. results, err := r.classifier.Classify(ctx, utterance) if err != nil { return Decision{}, err } best := results[0] // The runners-up are the interesting part: two intents a hundredth apart is // a different defect from one that won outright (V-564). Two are enough to // see that, and the rest of a seven-intent scoreboard is noise on the page. if rec := decision.From(ctx); rec != nil { for _, res := range results[1:min(len(results), 3)] { rec.Note(decision.Scored(decision.StageRoute, claimantClassifier, string(res.Intent), res.Score, decision.LostOnScore, "lower similarity than "+string(best.Intent))) } } // stage 2 — slot extraction for the winning intent. d := Decision{ Utterance: utterance, Stage: 2, Intent: best.Intent, Confidence: best.Score, Slots: r.extractor.Extract(ctx, best.Intent, utterance, now), Producer: RouteProducerClassifier, } // Capability selection: for the classifier path, the raw extractor // already filled Slots.Fn/Args via Extract(IntentAct). SelectCapability // records this as a deterministic bypass. If the raw match missed, // there is no LLM text to fall back to (Text == Utterance), so the // selection remains unresolved. applyCapabilityToSlots(&d, SelectCapability(d, r.extractor.Acts)) // stage 3 — confidence gate. Below threshold ⇒ clarify, don't guess. if d.Confidence < r.threshold { d.Stage = 3 d.Clarify = true } outcome, reason := decision.Won, "" if d.Clarify { outcome, reason = decision.Thinned, "below the clarify threshold, so she asks instead" } decision.Note(ctx, decision.Scored(decision.StageRoute, claimantClassifier, string(d.Intent), d.Confidence, outcome, reason)) return d, nil } func completeParsedSingleVerbFact(decision Decision) bool { return decision.Intent == IntentFact && decision.Slots.HasKey && completeSingleVerb(decision.Utterance) && !IsQuestionShaped(decision.Utterance) } // fillMatchedSlots — run stage-2 extraction over a decision some earlier // claimant produced, and fill only the slots that claimant left empty. A // matched value always wins: the claimant read the sentence, the extractor // guesses from keyword tables. // // Shared by the stage-0 grammars and the LLM router, which had the same hole // for the same reason. A grammar asserts an intent at confidence 1.0 and says // nothing about the slots, so "напомни в 11:00 позвонить маме" arrived with // HasTime false however plainly the hour was spoken, and the daemon read the // silence as absence and asked "Когда?" (V-572). The alternative was ten // grammars each re-implementing extraction. // // It is applied to every stage-0 decision rather than to a chosen few, because // for every intent but reminder it is inert: Extract fills Time for a reminder, // Fn for an act and Key for a fact, and nothing at all for query, system, note // or chat, which is what the query, clock, agenda, feed, list, task and // narrative rules emit. The act rules — wakeword-act and the Praxis ones — // already carry an Fn or they do not match, so there is nothing left for the // matcher to fill. The reminder rule is the one that gains, and its time parse // is a cost the daemon was already paying one layer down in actionReminder. // // Slots.Text is deliberately NOT filled here. Extract sets it to the raw // utterance, and a grammar that left it empty meant it: agendaQueryBuild hands // the query chain the utterance itself, and narrativeQueryBuild's Text is the // topic, not the sentence. // // If a reminder still has no time, leave it missing. The daemon then says it // could not read the time; inventing one would set a wrong alarm. // Returns what the extractor read, so a caller that wants more of it does not // pay for a second extraction — the reminder parser is the expensive one. func (r *Router) fillMatchedSlots(ctx context.Context, d *Decision, now time.Time) Slots { ex := r.extractor.Extract(ctx, d.Intent, d.Utterance, now) if !d.Slots.HasTime && ex.HasTime { d.Slots.Time, d.Slots.HasTime = ex.Time, ex.HasTime } if !d.Slots.HasKey && ex.HasKey { d.Slots.Key, d.Slots.Value, d.Slots.HasKey = ex.Key, ex.Value, ex.HasKey } // Capability selection (Fn/Args/HasFn) is now handled by // SelectCapability, not here. The extractor still fills them via // Extract(IntentAct), but the authoritative record lives in // Decision.CapabilitySelection. See capability.go. if !d.Slots.HasFn && ex.HasFn { d.Slots.Fn, d.Slots.Args, d.Slots.HasFn, d.Slots.ResolvedBy = ex.Fn, ex.Args, ex.HasFn, ex.ResolvedBy } return ex } // fillSlots — fillMatchedSlots for an LLM decision, plus the text backfill. // Capability selection (Fn/Args) is now handled by SelectCapability, called // after fillSlots in the routing pipeline. This function fills only time, key, // and text slots. func (r *Router) fillSlots(ctx context.Context, d *Decision, now time.Time) { ex := r.fillMatchedSlots(ctx, d, now) // The extractor's Text is the raw utterance, which is the payload for a // note, a query or a chat turn but not for a reminder — there Text is the // subject, what she says at the hour. Backfilling it made Text impossible // to be empty, so StillMissing never reported SlotText and "О чём // напомнить?" was unaskable; the answer to a question she did manage to // ask then overwrote the whole request instead of filling one gap // (Vikunja #383). A reminder with no subject stays empty and is gated // below into a question. if d.Slots.Text == "" && d.Intent != IntentReminder { d.Slots.Text = ex.Text } // Stage stays 1: it says who decided the route, and that was the LLM. } // gateLLMDecision — stage 3 for the LLM path (Vikunja #359). This used to be // the classifier's job alone (see the threshold check at the bottom of // Route): the LLM branch returned straight from fillSlots and never touched // r.threshold at all, so a hardcoded Confidence: 1.0 in llmrouter.go could // never gate. Two more structural holes are checked here, after fillSlots // has had a chance to fill them from the deterministic parsers — checking // before fillSlots would flag e.g. every keyless fact the fact parser goes // on to resolve (TestLLMFactGetsKeyFromParser): // - a fact with no key even after the parser tried — nothing to write, or // worse, a confident write under the wrong key; // - an act that never resolved to an allowlisted fn — a confident guess // here means either silently doing nothing or, if the daemon is lax, // running something never on the allowlist. Don't guess; ask. // // Anything below threshold gets the exact same Clarify=true treatment the // classifier path already produces — same field, same daemon-side consumer // (cmd/mavend/clarify.go), nothing new to wire. // // The act check now reads CapabilitySelection.Resolved (set by // SelectCapability before this function runs) rather than Slots.HasFn. func (r *Router) gateLLMDecision(d *Decision) { if d.Intent == IntentFact && !d.Slots.HasKey && d.Confidence > llmThinConfidence { d.Confidence = llmThinConfidence } if d.Intent == IntentAct && !d.CapabilitySelection.Resolved && d.Confidence > llmThinConfidence { d.Confidence = llmThinConfidence } // A reminder with no subject: she knows when but not what to say then. // Setting it anyway fires an empty reminder at the hour, which reads as a // bug to him and cannot be repaired after the fact. Ask (Vikunja #383). // // The test is what the text slot CONTAINS, not whether it is set. It was the // latter until 05-08-2026, and the slot is never empty: fillSlots hands it // the utterance, so "напомни" arrived here with Text:напомни and the gate // never fired (V-457). See remindersubject.go. if d.Intent == IntentReminder && !reminderHasSubject(d.Slots.Text) && d.Confidence > llmThinConfidence { d.Confidence = llmThinConfidence } if d.Confidence < r.threshold { d.Clarify = true } } // CorrectMisroute — the user corrected a bad classification. Appends a new // example for the corrected intent (append-only — grows the classifier, no // retrain). Same shape as nudges.outcome tuning cooldowns: more reliable over // time, introspectable, no model surgery. func (r *Router) CorrectMisroute(ctx context.Context, utterance string, corrected Intent) error { if r == nil || r.classifier == nil { // The LLM router can run with no classifier wired. The correction has // nowhere to land then, and the caller redoes the request anyway. return errors.New("router: no classifier to correct") } return r.classifier.AddExample(ctx, corrected, utterance) }