Files
Maven/internal/lexicon/lexicon.go
T
claude da9114b623 Preserve context across conversation intents (V-542)
Owner explicitly requested direct commits to master; bypass the branch-only hook.
2026-08-13 02:14:46 +04:00

463 lines
17 KiB
Go

// Package lexicon holds the Russian word sets that can be finished.
//
// A closed class has a fixed number of members: the language has as many
// interrogative pronouns as it has, and no utterance will ever contain a
// thirteenth month. Those sets belong in a data file, complete, and that is what
// this package is (Vikunja #522, owner's call 2026-08-04 — "not pattern, 100%").
//
// It is the first of three mechanisms that replaced hand-written Russian
// patterns, and the only one that answers with certainty. The other two are the
// embedder, for recognising an open set of phrasings, and a morphological
// dictionary, for questions about grammar. A list that can never be finished is
// a guess dressed as a rule and does not go here.
//
// What this package does NOT do is match. It hands out sets and lookups; the
// caller decides what to do with a hit, because "this token is an interrogative"
// and "this utterance is a question" are different claims.
package lexicon
import (
"embed"
"encoding/json"
"fmt"
"sort"
"strings"
"unicode"
"unicode/utf8"
)
//go:embed lexicon_ru_v1.json
var files embed.FS
// ruFile is the versioned file this package reads. A new version is a new file,
// not an edit to this one, so a caller pinned to v1 keeps the words it was
// measured against.
const ruFile = "lexicon_ru_v1.json"
type lexiconFile struct {
SchemaVersion int `json:"schema_version"`
Name string `json:"name"`
Sets map[string]lexiSet `json:"sets"`
}
type lexiSet struct {
Note string `json:"note"`
Words []string `json:"words"`
Values map[string]int `json:"values"`
}
// ru is parsed once at init. A malformed embedded file is a build-time mistake
// that survived to runtime, and there is no sane degraded behaviour for "the
// months are missing", so it panics rather than answering with an empty set.
var ru = mustLoad()
func mustLoad() lexiconFile {
data, err := files.ReadFile(ruFile)
if err != nil {
panic(fmt.Sprintf("lexicon: read %s: %v", ruFile, err))
}
var f lexiconFile
if err := json.Unmarshal(data, &f); err != nil {
panic(fmt.Sprintf("lexicon: parse %s: %v", ruFile, err))
}
for _, name := range []string{
"interrogatives", "capture_verbs", "narrative_requests", "cardinals", "ordinals",
"day_offsets", "weekdays", "weekdays_english", "months_genitive", "hours_spoken",
"not_place_after_v", "parts_of_day", "reminder_verbs", "half_hour",
"filler_particles", "task_done_words", "task_drop_words",
"confirm_yes", "confirm_no", "hour_units", "minute_units",
"conversation_verbs",
} {
s, ok := f.Sets[name]
if !ok || (len(s.Words) == 0 && len(s.Values) == 0) {
panic(fmt.Sprintf("lexicon: %s has no set %q", ruFile, name))
}
}
return f
}
// words returns a copy of a word set, so a caller cannot edit the lexicon by
// holding onto what it was given.
func words(set string) []string {
src := ru.Sets[set].Words
out := make([]string, len(src))
copy(out, src)
return out
}
// Interrogatives returns the question words, Russian and English.
func Interrogatives() []string { return words("interrogatives") }
// CaptureVerbs returns the imperatives that mean "record this".
func CaptureVerbs() []string { return words("capture_verbs") }
// NarrativeRequests returns the imperatives that mean "tell me about".
func NarrativeRequests() []string { return words("narrative_requests") }
// RepairMarkers lists the ways he says the previous turn was routed wrong. See
// the set's own note for why this one is a list and not a seed set.
func RepairMarkers() []string { return words("repair_markers") }
// RepairNegatives lists the ways he says the previous turn was wrong without
// saying what it should have been. Matched against the whole utterance, never as
// substrings — see the set's own note.
func RepairNegatives() []string { return words("repair_negatives") }
// FirstPerson lists every form of the first-person pronoun. Callers use it to
// decide that a sentence is about him: internal/router/complaint.go keeps a
// complaint out of the fact store unless one of these appears, because losing a
// fact he meant to store is the worse mistake.
func FirstPerson() []string { return words("first_person") }
// NotPlaceAfterV returns the words that follow "в" without naming a place.
func NotPlaceAfterV() []string { return words("not_place_after_v") }
// PartsOfDay returns the one-word names for a time of day: "вечером", "утром".
// They say which part of a day and never which day, so a caller that needs the
// day wants DayOffsetWords instead.
func PartsOfDay() []string { return words("parts_of_day") }
// ReminderVerbs returns the imperatives that open a reminder.
func ReminderVerbs() []string { return words("reminder_verbs") }
// Pleasantries returns the whole utterances that greet, thank or say goodbye.
// Whole utterances and not tokens: see the set's own note for why the tokens
// are unsafe alone.
func Pleasantries() []string { return words("pleasantries") }
// TaskDoneWords returns the words that finish a task, and TaskDropWords the
// words that abandon one. Two sets rather than one with a value, because the
// store records which of the two happened and the caller has to say so.
//
// Both mix moods on purpose, and the caller must match them the way the sets'
// notes say: an imperative exactly, a stative by lemma.
func TaskDoneWords() []string { return words("task_done_words") }
// ConfirmYes returns the words that answer a parked confirm with yes, and
// ConfirmNo the ones that answer it with no. Some members are multi-word ("не
// надо"), so a caller matches longest-first over tokens rather than looking up
// one word at a time. See the sets' notes for why neither may be matched as a
// substring.
func ConfirmYes() []string { return words("confirm_yes") }
// ConfirmNo — see ConfirmYes.
func ConfirmNo() []string { return words("confirm_no") }
// TaskDropWords — see TaskDoneWords.
func TaskDropWords() []string { return words("task_drop_words") }
// WaterNouns and DrinkVerbs are the two halves of a water fact: he has to name
// the drink and the drinking, because "вода" alone is a word about water and
// "выпил" alone does not say what. The other four self-care sets need only one
// word each. All six are matched over tokens by lemma — except ShowerWords, see
// its own note.
func WaterNouns() []string { return words("water_nouns") }
// DrinkVerbs — see WaterNouns.
func DrinkVerbs() []string { return words("drink_verbs") }
// MealWords returns the nouns and verbs of having eaten.
func MealWords() []string { return words("meal_words") }
// ShowerWords returns the shower noun. Match these EXACTLY and not by lemma:
// the dictionary makes "душ" and "душа" one word, and only one of them is a
// shower. The set's note says why exact matching costs nothing here.
func ShowerWords() []string { return words("shower_words") }
// BreakWords returns the noun and verbs of taking a break.
func BreakWords() []string { return words("break_words") }
// SleepWords returns the verbs of having slept.
func SleepWords() []string { return words("sleep_words") }
// SlotValueFrame returns the words that can surround a bare slot value without
// making the utterance a request of its own. A caller strips these (along with
// the numbers and the other closed time sets) to see whether an utterance
// carries any content beside the value it was asked for. See the set's note.
// The hour and the minute nouns are part of the frame and are kept in their own
// sets, so there is one copy of each closed class rather than a copy per caller.
func SlotValueFrame() []string {
out := words("slot_value_frame")
out = append(out, HourUnits()...)
out = append(out, MinuteUnits()...)
return out
}
// HourUnits returns every form of the hour noun, and MinuteUnits every form of
// the minute noun. One home for each, because four router sets used to list the
// hour and all four stopped at "часу" (V-609). A caller folding time words into
// one set reads these; a caller asking about a single word reads IsHourUnit or
// IsMinuteUnit.
func HourUnits() []string { return words("hour_units") }
// MinuteUnits — see HourUnits.
func MinuteUnits() []string { return words("minute_units") }
// IsHourUnit reports whether a word is the hour noun in any form.
func IsHourUnit(word string) bool { return inSet("hour_units", word) }
// IsMinuteUnit reports whether a word is the minute noun in any form.
func IsMinuteUnit(word string) bool { return inSet("minute_units", word) }
func inSet(set, word string) bool {
w := norm(word)
for _, s := range ru.Sets[set].Words {
if w == s {
return true
}
}
return false
}
// DialogueCancel returns the ways he calls off the request Maven is assembling.
// Distinct from TaskDropWords, which abandons an item that already exists.
func DialogueCancel() []string { return words("dialogue_cancel") }
// ConversationVerbs returns Maven's vocabulary for explicitly opening a
// conversation after a cooperative marker ("давай поговорим", "let's talk").
// The caller matches Russian members by lemma and English members exactly.
// Keeping the verbs here prevents session state from growing its own stem list.
func ConversationVerbs() []string { return words("conversation_verbs") }
// IsFillerParticle reports whether a word can never be the subject of a
// request: a particle, a politeness word, or the first-person object. See the
// set's own note for why this is not a stopword list.
func IsFillerParticle(word string) bool {
w := norm(word)
for _, p := range ru.Sets["filler_particles"].Words {
if w == p {
return true
}
}
return false
}
// HalfHourWords returns those forms, for a caller folding every time word into
// one set rather than asking about one word.
func HalfHourWords() []string { return words("half_hour") }
// IsHalfHour reports whether a word introduces a spoken half hour, so the
// ordinal after it is an hour rather than a position. One caller reads that
// ordinal as the hour and another has to decline it; both ask here.
func IsHalfHour(word string) bool {
w := norm(word)
for _, h := range ru.Sets["half_hour"].Words {
if w == h {
return true
}
}
return false
}
// Cardinal reports the value of a spoken number word. The word is compared
// lowercased and trimmed, because it arrives from a tokenizer that may not have
// done either.
func Cardinal(word string) (int, bool) {
n, ok := ru.Sets["cardinals"].Values[norm(word)]
return n, ok
}
// Ordinal reports the 1-based position a position word names, with -1 for the
// last one. Same lookup shape as Cardinal, and the same reason: "второй" and
// "вторым" are one position, and a caller matching stems would also match
// "вторник".
func Ordinal(word string) (int, bool) {
n, ok := ru.Sets["ordinals"].Values[norm(word)]
return n, ok
}
// Ordinals returns the position words with their positions, sorted, so a caller
// that needs a form this set does not list can ask a morphological dictionary
// whether one of these is the same word. Sorted because map order is not stable
// and a caller folding these into a pattern would otherwise build a different one
// every run.
func Ordinals() []struct {
Word string
N int
} {
vals := ru.Sets["ordinals"].Values
out := make([]struct {
Word string
N int
}, 0, len(vals))
for w, n := range vals {
out = append(out, struct {
Word string
N int
}{w, n})
}
sort.Slice(out, func(i, j int) bool { return out[i].Word < out[j].Word })
return out
}
// OrdinalIn reports the position word that comes FIRST in a sentence, so a
// caller does not have to tokenize before asking. Word-boundary matched for the
// reason above, and earliest-wins rather than first-found: map iteration order
// would otherwise answer "отметь первый и второй" differently between runs.
func OrdinalIn(text string) (int, bool) {
lower := norm(text)
best, at := 0, -1
for w, n := range ru.Sets["ordinals"].Values {
i := indexWord(lower, w)
if i < 0 || (at >= 0 && i > at) {
continue
}
// Two different words cannot match at one offset: both ends are
// boundary-checked, so no key is a prefix of another as matched.
best, at = n, i
}
return best, at >= 0
}
// DayOffset reports how many days a relative day word moves from today.
//
// The zero value is a real answer here — "сегодня" is offset 0 — so the second
// return is the only way to tell a hit from a miss. Callers that used to switch
// on strings.Contains had to order "послезавтра" before "завтра" by hand,
// because one contains the other; a lookup has no such trap.
func DayOffset(word string) (int, bool) {
n, ok := ru.Sets["day_offsets"].Values[norm(word)]
return n, ok
}
// DayOffsetWords lists the relative day words themselves, sorted so the order is
// stable across builds — a caller that folds them into a regexp alternation would
// otherwise produce a different pattern every run. Map iteration order is why
// this sorts rather than the caller.
func DayOffsetWords() []string {
vals := ru.Sets["day_offsets"].Values
out := make([]string, 0, len(vals))
for w := range vals {
out = append(out, w)
}
sort.Strings(out)
return out
}
// DayOffsetIn finds a relative day word anywhere in a phrase and reports its
// offset. Where two words appear, the one that moves furthest from today wins in
// absolute terms: "не сегодня, а послезавтра" is about the day after tomorrow,
// and the longest-match rule that picking the first word would need is exactly
// what the old Contains switch got wrong.
func DayOffsetIn(text string) (int, bool) {
lower := norm(text)
best, found := 0, false
for word, n := range ru.Sets["day_offsets"].Values {
if !containsWord(lower, word) {
continue
}
if !found || abs(n) > abs(best) {
best, found = n, true
}
}
return best, found
}
// Weekday returns the Russian name of a weekday index, Sunday first, matching
// Go's time.Weekday. An index off the end returns "".
func Weekday(i int) string { return at("weekdays", i) }
// Weekdays returns the seven Russian names in one slice, Sunday first, for a
// caller matching a token against all of them rather than rendering one. Only
// the nominative is here: every other case lemmatises to it, so an oblique form
// is morph's question and not a second list (V-581).
func Weekdays() []string { return words("weekdays") }
// WeekdayEnglish reports the Go time.Weekday index an English weekday names,
// singular or plural. English needs the list that Russian does not, because the
// vendored dictionary is Russian and leaves "mondays" as it found it.
func WeekdayEnglish(word string) (int, bool) {
n, ok := ru.Sets["weekdays_english"].Values[norm(word)]
return n, ok
}
// MonthGenitive returns the month name a date takes — "10 июля", not "июль".
// The set is 1-indexed, so MonthGenitive(int(t.Month())) is the whole call.
func MonthGenitive(m int) string { return at("months_genitive", m) }
// HourSpoken returns an hour spelled out for the voice. 0 to 23.
func HourSpoken(h int) string { return at("hours_spoken", h) }
func at(set string, i int) string {
w := ru.Sets[set].Words
if i < 0 || i >= len(w) {
return ""
}
return w[i]
}
func norm(s string) string { return strings.ToLower(strings.TrimSpace(s)) }
func abs(n int) int {
if n < 0 {
return -n
}
return n
}
// indexWord is containsWord returning where the match starts, or -1.
func indexWord(haystack, needle string) int {
if needle == "" {
return -1
}
from := 0
for {
i := strings.Index(haystack[from:], needle)
if i < 0 {
return -1
}
i += from
if boundaryBefore(haystack, i) && boundaryAfter(haystack, i+len(needle)) {
return i
}
from = i + len(needle)
if from >= len(haystack) {
return -1
}
}
}
// containsWord reports whether haystack holds needle on word boundaries. Go's
// \b is ASCII-only and never fires after a Cyrillic letter, so the boundary is
// checked here instead: a rune on either side must not be a letter or a digit.
func containsWord(haystack, needle string) bool {
if needle == "" {
return false
}
from := 0
for {
i := strings.Index(haystack[from:], needle)
if i < 0 {
return false
}
i += from
if boundaryBefore(haystack, i) && boundaryAfter(haystack, i+len(needle)) {
return true
}
from = i + len(needle)
if from >= len(haystack) {
return false
}
}
}
func boundaryBefore(s string, i int) bool {
if i == 0 {
return true
}
r, _ := utf8.DecodeLastRuneInString(s[:i])
return !wordRune(r)
}
func boundaryAfter(s string, i int) bool {
if i >= len(s) {
return true
}
r, _ := utf8.DecodeRuneInString(s[i:])
return !wordRune(r)
}
func wordRune(r rune) bool {
return unicode.IsLetter(r) || unicode.IsDigit(r) || r == '_'
}