// Command labelgen labels utterances with the stage 0 grammars and prints JSONL. // // docs/plans/18-routing-heads-on-e5-small.md calls the labeled set the whole // project, and it names the stage 0 grammars as the high-precision label // functions to start from. This runs them — the real ones, in the real // buildRouter order — rather than a reimplementation, so a rule change moves // the training data with it. // // A grammar that declines leaves the line unlabeled. Those go to the model, and // keeping them is the point: a set labeled only by the rules teaches only the // rules. // // go run ./cmd/labelgen < utterances.txt > labeled.jsonl // // The wakeword-act grammar is absent, because its allowlist is the deployment's // enabled tool names and this tool has no deployment. Every other rule is here. package main import ( "bufio" "encoding/json" "fmt" "os" "strings" "github.com/kami/maven/internal/router" ) // label is one output row. The grammar name rides along so a reviewer can see // which rule made the claim, and so a rule that turns out to be wrong can have // its rows pulled without re-running everything. type label struct { Utterance string `json:"utterance"` Intent string `json:"intent,omitempty"` Grammar string `json:"grammar,omitempty"` Key string `json:"key,omitempty"` Value string `json:"value,omitempty"` Fn string `json:"fn,omitempty"` Text string `json:"text,omitempty"` Labeled bool `json:"labeled"` } // grammars mirrors buildRouter's order in cmd/mavend/voicewire.go. Order is // load-bearing there and so it is here: the agenda rules must sit after the // clock rules, Praxis before the capture marker, the narrative rules last. func grammars() []router.Grammar { var g []router.Grammar g = append(g, router.SystemTimeDateGrammars()...) g = append(g, router.AgendaQueryGrammars()...) g = append(g, router.FeedQueryGrammar()) g = append(g, router.TaskListGrammar()) g = append(g, router.ListGrammars()...) g = append(g, router.ReminderGrammar()) g = append(g, router.PraxisGrammars()...) g = append(g, router.TaskCaptureGrammar()) g = append(g, router.NarrativeQueryGrammars()...) return g } func match(gs []router.Grammar, utterance string) label { out := label{Utterance: utterance} for _, g := range gs { m := g.Pattern.FindStringSubmatch(utterance) if m == nil { continue } d, ok := g.Build(m) if !ok { continue // the rule saw its shape and declined it } out.Intent = string(d.Intent) out.Grammar = g.Name out.Key = d.Slots.Key out.Value = d.Slots.Value out.Fn = d.Slots.Fn out.Text = d.Slots.Text out.Labeled = true return out } return out } func main() { gs := grammars() in := bufio.NewScanner(os.Stdin) in.Buffer(make([]byte, 0, 64*1024), 1024*1024) out := bufio.NewWriter(os.Stdout) defer out.Flush() enc := json.NewEncoder(out) var seen, labeled int for in.Scan() { line := strings.TrimSpace(in.Text()) if line == "" || strings.HasPrefix(line, "#") { continue } seen++ l := match(gs, line) if l.Labeled { labeled++ } if err := enc.Encode(l); err != nil { fmt.Fprintln(os.Stderr, "labelgen:", err) os.Exit(1) } } if err := in.Err(); err != nil { fmt.Fprintln(os.Stderr, "labelgen:", err) os.Exit(1) } // Coverage on stderr, so the count is visible without polluting the JSONL. fmt.Fprintf(os.Stderr, "labelgen: %d/%d labeled by %d grammars\n", labeled, seen, len(gs)) }