Compare commits
22 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0b994ff1c3 | |||
| 0b1efe4911 | |||
| 580959f856 | |||
| bf2587c7fa | |||
| 26ff646ace | |||
| 9e1958e7b0 | |||
| e6923490fd | |||
| d7a43afd90 | |||
| fabc3bc274 | |||
| b6eed20af2 | |||
| 936c6d71db | |||
| 72aa97dae8 | |||
| 1c0a1d0db0 | |||
| 37feee1eb3 | |||
| 94c273780a | |||
| 93c08f9de1 | |||
| 4f96bbd6ec | |||
| 7dba1b7935 | |||
| 8102c73f83 | |||
| 8c36e7ef84 | |||
| 95cbf82e38 | |||
| 5bca435146 |
+14
-9
@@ -17,6 +17,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/lexicon"
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/say"
|
||||
)
|
||||
|
||||
@@ -73,22 +74,26 @@ func mentionsUnknownPlace(u string) bool {
|
||||
// date for a day she did not understand.
|
||||
const onlyNearDaysReply = "я считаю только сегодня, завтра, послезавтра и вчера — про другие дни пока не скажу."
|
||||
|
||||
// dayWords — day references the calendar parser cannot resolve. A weekday name
|
||||
// or a "через …" phrase means he asked about a specific other day.
|
||||
var dayWords = []string{
|
||||
"понедельник", "вторник", "сред", "четверг", "пятниц", "суббот", "воскресен",
|
||||
"через", "monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday",
|
||||
}
|
||||
|
||||
// mentionsUnknownDay reports whether the question names a day the calendar
|
||||
// parser could not resolve. Mirror of mentionsUnknownPlace: it exists only to
|
||||
// pick an honest reply over a confidently wrong one.
|
||||
//
|
||||
// Only called after ParseCalendarDate has already failed, so "завтра" and the
|
||||
// other words it does know never reach here.
|
||||
//
|
||||
// The weekday half was a list of STEMS matched with strings.Contains until
|
||||
// V-581 — "сред", "пятниц", "суббот". That is the hand-written Russian pattern
|
||||
// the sweep of 2026-08-04 took out, and it was wrong in the way such a pattern
|
||||
// always is: "среди", "средство" and "средний" all contain "сред", so a question
|
||||
// carrying any of them was answered with onlyNearDaysReply instead of the date.
|
||||
// Whole tokens now, and the weekday itself is router.WeekdayIndex, which reads
|
||||
// the lexicon and asks the dictionary about the case.
|
||||
func mentionsUnknownDay(u string) bool {
|
||||
for _, w := range dayWords {
|
||||
if strings.Contains(u, w) {
|
||||
for _, tok := range quietTokens(u) {
|
||||
if tok == "через" {
|
||||
return true
|
||||
}
|
||||
if _, ok := router.WeekdayIndex(tok); ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
package main
|
||||
|
||||
import "testing"
|
||||
|
||||
// TestMentionsUnknownDayReadsWordsNotStems — the defect V-581 found. The
|
||||
// weekday half of this guard was a list of stems matched with strings.Contains,
|
||||
// so "среди", "средство" and "средний" all read as Wednesday and the question
|
||||
// was answered with onlyNearDaysReply instead of a date.
|
||||
//
|
||||
// The other half of the fix is coverage: a stem list stops at the forms whoever
|
||||
// wrote it thought of, and "воскресеньях" was not one of them.
|
||||
func TestMentionsUnknownDayReadsWordsNotStems(t *testing.T) {
|
||||
for _, u := range []string{
|
||||
"какое число в понедельник",
|
||||
"какое число в среду",
|
||||
"какое число в среде",
|
||||
"что там по воскресеньям",
|
||||
"what is the date on friday",
|
||||
"какое число через неделю",
|
||||
} {
|
||||
if !mentionsUnknownDay(u) {
|
||||
t.Errorf("mentionsUnknownDay(%q) = false, want true", u)
|
||||
}
|
||||
}
|
||||
for _, u := range []string{
|
||||
"какое число в среднем",
|
||||
"сколько это в среднем",
|
||||
"какое сегодня средство",
|
||||
"какое число",
|
||||
} {
|
||||
if mentionsUnknownDay(u) {
|
||||
t.Errorf("mentionsUnknownDay(%q) = true; it names no day", u)
|
||||
}
|
||||
}
|
||||
}
|
||||
+39
-9
@@ -7,6 +7,10 @@ package main
|
||||
import (
|
||||
"regexp"
|
||||
"strings"
|
||||
|
||||
"github.com/kami/maven/internal/lexicon"
|
||||
"github.com/kami/maven/internal/morph"
|
||||
"github.com/kami/maven/internal/router"
|
||||
)
|
||||
|
||||
// isWeatherQuery returns true if the utterance is about weather.
|
||||
@@ -27,16 +31,42 @@ func isWeatherQuery(u string) bool {
|
||||
// come through whole and "в 5 утра" does not.
|
||||
var weatherPlace = regexp.MustCompile(`(?i)(?:^|\s)(?:в|во|in)\s+([\p{L}-]+(?:\s+[\p{L}-]+)?)`)
|
||||
|
||||
// weatherNonPlaces — words that follow "в" in a weather question and are not
|
||||
// cities. "какая погода в доме" is the smart-home sensor, not Open-Meteo, and
|
||||
// "тепло в комнате" is the same question about the same room.
|
||||
var weatherNonPlaces = map[string]bool{
|
||||
// weatherRooms — the rooms of the house, which are the only words in this
|
||||
// guard that belong to it. "какая погода в доме" is the smart-home sensor, not
|
||||
// Open-Meteo, and "тепло в комнате" is the same question about the same room.
|
||||
//
|
||||
// The rest of the guard used to be a third copy of three closed sets that
|
||||
// already exist in the lexicon: the weekdays, the parts of the day, and the
|
||||
// words that follow "в" without naming a place (V-581). Each copy was short in
|
||||
// its own direction — "среду" but not "среде", "утром" but not "утра", "целом"
|
||||
// but not "общем" — so the same question phrased one word differently reached
|
||||
// the geocoder as a city.
|
||||
var weatherRooms = map[string]bool{
|
||||
"доме": true, "квартире": true, "комнате": true, "спальне": true,
|
||||
"гостиной": true, "кухне": true, "гараже": true, "офисе": true,
|
||||
"выходные": true, "субботу": true, "воскресенье": true, "понедельник": true,
|
||||
"вторник": true, "среду": true, "четверг": true, "пятницу": true,
|
||||
"обед": true, "обеде": true, "утро": true, "утром": true, "вечер": true,
|
||||
"вечером": true, "ночь": true, "ночью": true, "целом": true, "принципе": true,
|
||||
"обед": true, "обеде": true, "выходные": true, "выходных": true,
|
||||
}
|
||||
|
||||
// isWeatherNonPlace reports whether the word after "в" names something other
|
||||
// than a place he could ask the weather for.
|
||||
func isWeatherNonPlace(word string) bool {
|
||||
if weatherRooms[word] {
|
||||
return true
|
||||
}
|
||||
if _, ok := router.WeekdayIndex(word); ok {
|
||||
return true
|
||||
}
|
||||
for _, w := range lexicon.PartsOfDay() {
|
||||
if word == w || morph.SameWord(word, w) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
for _, w := range lexicon.NotPlaceAfterV() {
|
||||
if word == w {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// extractWeatherLocation returns the place he named, or the configured default
|
||||
@@ -63,7 +93,7 @@ func extractWeatherLocation(u, defaultLoc string) string {
|
||||
}
|
||||
place := strings.TrimSpace(m[1])
|
||||
first := strings.ToLower(strings.Fields(place)[0])
|
||||
if weatherNonPlaces[first] {
|
||||
if isWeatherNonPlace(first) {
|
||||
return defaultLoc
|
||||
}
|
||||
return place
|
||||
|
||||
@@ -29,6 +29,13 @@ func TestExtractWeatherLocation(t *testing.T) {
|
||||
// the house sensors and the day words answer elsewhere.
|
||||
{"тепло в комнате?", "Berlin", "Berlin"},
|
||||
{"какая погода в выходные", "Berlin", "Berlin"},
|
||||
// The cases the three private copies of the lexicon were short by
|
||||
// (V-581): a weekday in a case the old map did not list, a part of the
|
||||
// day in one it did not list, and "в общем".
|
||||
{"какая погода в среде", "Berlin", "Berlin"},
|
||||
{"какая погода в воскресеньях", "Berlin", "Berlin"},
|
||||
{"какая погода в понедельникам", "Berlin", "Berlin"},
|
||||
{"какая погода в общем", "Berlin", "Berlin"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := extractWeatherLocation(c.utterance, c.def); got != c.want {
|
||||
|
||||
@@ -82,6 +82,38 @@ func TestWAVRoundTrip(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A LIST chunk sitting between fmt and data is common (arecord and ffmpeg both
|
||||
// write one), and its payload is free text that can spell "data". The parser
|
||||
// walks chunk headers, so the text is skipped and the real samples are read.
|
||||
func TestPCMFromWAVSkipsLISTChunk(t *testing.T) {
|
||||
t.Parallel()
|
||||
pcm := []byte{1, 0, 2, 0, 3, 0, 4, 0}
|
||||
list := []byte("LIST")
|
||||
payload := []byte("INFOICMTdata is not here")
|
||||
list = binary.LittleEndian.AppendUint32(list, uint32(len(payload)))
|
||||
list = append(list, payload...)
|
||||
|
||||
plain, err := WAVFromPCM(PCM16kMono, pcm)
|
||||
if err != nil {
|
||||
t.Fatalf("WAVFromPCM: %v", err)
|
||||
}
|
||||
wav := append([]byte{}, plain[:36]...)
|
||||
wav = append(wav, list...)
|
||||
wav = append(wav, plain[36:]...)
|
||||
binary.LittleEndian.PutUint32(wav[4:8], uint32(len(wav)-8))
|
||||
|
||||
f, got, err := PCMFromWAV(wav)
|
||||
if err != nil {
|
||||
t.Fatalf("PCMFromWAV: %v", err)
|
||||
}
|
||||
if !f.IsValid() {
|
||||
t.Fatalf("parsed format invalid: %+v", f)
|
||||
}
|
||||
if !bytes.Equal(got, pcm) {
|
||||
t.Fatalf("PCM mismatch: got %v, want %v", got, pcm)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPCMFromWAVRejectsNonCanonical(t *testing.T) {
|
||||
t.Parallel()
|
||||
// too short
|
||||
|
||||
+30
-12
@@ -36,6 +36,10 @@ const wavHeaderSize = 44
|
||||
// raw PCM samples (little-endian int16 as bytes). A non-canonical blob is
|
||||
// rejected with ErrNotCanonicalPCM; the format mismatch is logged at the seam
|
||||
// so the caller surfaces it, not a hidden silent downmix.
|
||||
//
|
||||
// The returned PCM aliases wav rather than copying it, because a recording is
|
||||
// large and the caller already owns the bytes. A caller that keeps the PCM past
|
||||
// the life of wav, or that reuses wav as a read buffer, must copy first.
|
||||
func PCMFromWAV(wav []byte) (Format, []byte, error) {
|
||||
if len(wav) < wavHeaderSize {
|
||||
return Format{}, nil, fmt.Errorf("audio: wav too short: %d bytes", len(wav))
|
||||
@@ -61,17 +65,13 @@ func PCMFromWAV(wav []byte) (Format, []byte, error) {
|
||||
return Format{}, nil, fmt.Errorf("%w: channels=%d bits=%d (want 1/16)", ErrNotCanonicalPCM, channels, bitsPerSample)
|
||||
}
|
||||
// data chunk: the spec mandates it appears right after fmt, but real
|
||||
// recorders sometimes append extra chunks (LIST, fact). Find the "data"
|
||||
// chunk by scanning; require it within the region we'd expect.
|
||||
dataIdx := -1
|
||||
for i := wavHeaderSize - 8; i+8 <= len(wav) && i < wavHeaderSize+4096; i++ {
|
||||
if string(wav[i:i+4]) == "data" {
|
||||
dataIdx = i
|
||||
break
|
||||
}
|
||||
}
|
||||
if dataIdx < 0 {
|
||||
return Format{}, nil, fmt.Errorf("%w: no data chunk", ErrNotCanonicalPCM)
|
||||
// recorders sometimes append extra chunks (LIST, fact). Walk the chunk
|
||||
// headers rather than scanning for the four bytes "data", because those
|
||||
// bytes occur inside a LIST/INFO payload as ordinary text and a byte scan
|
||||
// would take the middle of a comment for a chunk header.
|
||||
dataIdx, err := findDataChunk(wav)
|
||||
if err != nil {
|
||||
return Format{}, nil, err
|
||||
}
|
||||
dataSize := binary.LittleEndian.Uint32(wav[dataIdx+4 : dataIdx+8])
|
||||
body := wav[dataIdx+8:]
|
||||
@@ -90,6 +90,24 @@ func PCMFromWAV(wav []byte) (Format, []byte, error) {
|
||||
return f, body, nil
|
||||
}
|
||||
|
||||
// findDataChunk returns the offset of the "data" chunk header, walking the
|
||||
// chunk list that starts after the 16-byte fmt chunk. Chunks are word-aligned,
|
||||
// so an odd size carries one pad byte the next header sits behind.
|
||||
func findDataChunk(wav []byte) (int, error) {
|
||||
for pos := wavHeaderSize - 8; pos+8 <= len(wav); {
|
||||
size := int(binary.LittleEndian.Uint32(wav[pos+4 : pos+8]))
|
||||
if string(wav[pos:pos+4]) == "data" {
|
||||
return pos, nil
|
||||
}
|
||||
next := pos + 8 + size + size%2
|
||||
if next <= pos || next > len(wav) {
|
||||
break
|
||||
}
|
||||
pos = next
|
||||
}
|
||||
return 0, fmt.Errorf("%w: no data chunk", ErrNotCanonicalPCM)
|
||||
}
|
||||
|
||||
// WAVFromPCM wraps raw 16-bit mono PCM bytes in a canonical 44-byte WAV
|
||||
// header so the result can be written to disk and played with `aplay`.
|
||||
// Used by the reference client to write the TTS reply; not on the wire.
|
||||
@@ -115,7 +133,7 @@ const WAVHeaderSize = wavHeaderSize
|
||||
// avoiding.
|
||||
func WAVHeader(format Format, n int) ([]byte, error) {
|
||||
if !format.IsValid() {
|
||||
return nil, fmt.Errorf("audio: WAVFromPCM: %w: %+v", ErrNotCanonicalPCM, format)
|
||||
return nil, fmt.Errorf("audio: WAVHeader: %w: %+v", ErrNotCanonicalPCM, format)
|
||||
}
|
||||
out := make([]byte, wavHeaderSize)
|
||||
// RIFF header
|
||||
|
||||
@@ -140,6 +140,12 @@ const EventKeyPrefix = "calendar_event_"
|
||||
//
|
||||
// An end at or before the start is read as crossing midnight, so a 23:30-00:15
|
||||
// meeting covers the quarter hour it actually covers.
|
||||
//
|
||||
// Both readings are built with time.Date rather than added to midnight as a
|
||||
// duration. A day is 23 or 25 hours wide on the two DST changeovers, so
|
||||
// midnight plus fourteen hours is 13:00 or 15:00 on those days, and the busy
|
||||
// gate would then read a 14:00 meeting an hour off. The same goes for the
|
||||
// midnight crossing, which is AddDate and not a 24-hour add.
|
||||
func FactSpan(key, value string, loc *time.Location) (start, end time.Time, ok bool) {
|
||||
if !strings.HasPrefix(key, EventKeyPrefix) {
|
||||
return time.Time{}, time.Time{}, false
|
||||
@@ -172,10 +178,11 @@ func FactSpan(key, value string, loc *time.Location) (start, end time.Time, ok b
|
||||
if !ok1 || !ok2 {
|
||||
return time.Time{}, time.Time{}, false
|
||||
}
|
||||
start = day.Add(time.Duration(sh)*time.Hour + time.Duration(sm)*time.Minute)
|
||||
end = day.Add(time.Duration(eh)*time.Hour + time.Duration(em)*time.Minute)
|
||||
y, mo, d := day.Date()
|
||||
start = time.Date(y, mo, d, sh, sm, 0, 0, loc)
|
||||
end = time.Date(y, mo, d, eh, em, 0, 0, loc)
|
||||
if !end.After(start) {
|
||||
end = end.Add(24 * time.Hour)
|
||||
end = end.AddDate(0, 0, 1)
|
||||
}
|
||||
return start, end, true
|
||||
}
|
||||
|
||||
@@ -66,7 +66,7 @@ func ParseICalDay(body []byte, now time.Time) []Event {
|
||||
// Reports false for all-day events and parse failures.
|
||||
func parseVEVENT(block string, loc *time.Location) (Event, bool) {
|
||||
var e Event
|
||||
for _, line := range strings.Split(block, "\n") {
|
||||
for _, line := range strings.Split(unfold(block), "\n") {
|
||||
line = strings.TrimSpace(line)
|
||||
switch {
|
||||
case strings.HasPrefix(line, "DTSTART"):
|
||||
@@ -78,9 +78,9 @@ func parseVEVENT(block string, loc *time.Location) (Event, bool) {
|
||||
e.End = t
|
||||
}
|
||||
case strings.HasPrefix(line, "SUMMARY"):
|
||||
e.Summary = afterColon(line)
|
||||
e.Summary = unescapeText(afterColon(line))
|
||||
case strings.HasPrefix(line, "UID"):
|
||||
e.UID = afterColon(line)
|
||||
e.UID = unescapeText(afterColon(line))
|
||||
}
|
||||
}
|
||||
if e.Start.IsZero() || e.End.IsZero() {
|
||||
@@ -89,6 +89,48 @@ func parseVEVENT(block string, loc *time.Location) (Event, bool) {
|
||||
return e, true
|
||||
}
|
||||
|
||||
// unfold undoes RFC 5545 content-line folding, where a long property is split
|
||||
// with a CRLF and the continuation begins with one space or tab.
|
||||
//
|
||||
// It runs before the block is split into lines, because splitting first and
|
||||
// trimming each line destroys the leading space that marks a continuation. A
|
||||
// server folds at 75 octets and a Russian summary is two bytes a letter, so
|
||||
// "Еженедельная планёрка с командой" crosses the limit easily — without this
|
||||
// the tail of the summary was read as an unknown property and dropped, and the
|
||||
// event was filed under a truncated name.
|
||||
func unfold(block string) string {
|
||||
if !strings.Contains(block, "\n ") && !strings.Contains(block, "\n\t") {
|
||||
return block
|
||||
}
|
||||
return strings.NewReplacer("\r\n ", "", "\r\n\t", "", "\n ", "", "\n\t", "").Replace(block)
|
||||
}
|
||||
|
||||
// unescapeText reverses the RFC 5545 TEXT escaping escapeText applies. Without
|
||||
// it a summary a server wrote as "Обед\, потом созвон" reaches the day plan
|
||||
// with the backslash still in it, and FactKey folds that literal into the key.
|
||||
func unescapeText(s string) string {
|
||||
if !strings.Contains(s, `\`) {
|
||||
return s
|
||||
}
|
||||
var b strings.Builder
|
||||
b.Grow(len(s))
|
||||
for i := 0; i < len(s); i++ {
|
||||
if s[i] != '\\' || i+1 >= len(s) {
|
||||
b.WriteByte(s[i])
|
||||
continue
|
||||
}
|
||||
i++
|
||||
switch s[i] {
|
||||
case 'n', 'N':
|
||||
b.WriteByte('\n')
|
||||
default:
|
||||
// ";", ",", "\\" and anything else a writer escaped needlessly.
|
||||
b.WriteByte(s[i])
|
||||
}
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func afterColon(line string) string {
|
||||
if i := strings.Index(line, ":"); i >= 0 {
|
||||
return strings.TrimSpace(line[i+1:])
|
||||
|
||||
@@ -61,6 +61,44 @@ func TestRenderICalEscapesInjection(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A folded SUMMARY is one property, not a property plus a dropped tail. Servers
|
||||
// fold at 75 octets and a Russian summary is two bytes a letter.
|
||||
func TestParseICalUnfoldsAndUnescapes(t *testing.T) {
|
||||
body := []byte("BEGIN:VEVENT\r\n" +
|
||||
"UID:u1\r\n" +
|
||||
"DTSTART:20260703T130000Z\r\n" +
|
||||
"DTEND:20260703T140000Z\r\n" +
|
||||
"SUMMARY:Еженедельная планёрка\\, потом\r\n созвон\r\n" +
|
||||
"END:VEVENT\r\n")
|
||||
from := time.Date(2026, 7, 3, 0, 0, 0, 0, time.UTC)
|
||||
events := ParseICal(body, from, from.AddDate(0, 0, 1))
|
||||
if len(events) != 1 {
|
||||
t.Fatalf("got %d events, want 1", len(events))
|
||||
}
|
||||
if want := "Еженедельная планёрка, потом созвон"; events[0].Summary != want {
|
||||
t.Errorf("Summary = %q, want %q", events[0].Summary, want)
|
||||
}
|
||||
}
|
||||
|
||||
// A day is 23 hours wide where DST starts, so a wall clock reading has to be
|
||||
// built with time.Date and never as midnight plus a duration.
|
||||
func TestFactSpanAcrossDSTStart(t *testing.T) {
|
||||
loc, err := time.LoadLocation("Europe/Berlin")
|
||||
if err != nil {
|
||||
t.Skipf("no tzdata for Europe/Berlin: %v", err)
|
||||
}
|
||||
start, end, ok := FactSpan("calendar_event_20260329_Planerka", "Planerka @ 14:00-15:00", loc)
|
||||
if !ok {
|
||||
t.Fatal("FactSpan reported not ok")
|
||||
}
|
||||
if start.Hour() != 14 || start.Minute() != 0 {
|
||||
t.Errorf("start = %s, want a 14:00 wall clock", start)
|
||||
}
|
||||
if end.Hour() != 15 {
|
||||
t.Errorf("end = %s, want a 15:00 wall clock", end)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReminderEventEmptyPayload(t *testing.T) {
|
||||
e := ReminderEvent(3, time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC), " ", 0)
|
||||
if e.Summary != "напоминание" {
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
// utterance (V-565, umbrella V-558, design in
|
||||
// docs/plans/19-dialogue-arbitration.md).
|
||||
//
|
||||
// Maven's cascade has roughly ten stage-0 grammars, seven router intents,
|
||||
// twenty-two query sources and four stateful pre-emptors, and every one of them
|
||||
// Maven's cascade has twenty-two stage-0 grammars, seven router intents,
|
||||
// twenty-two query sources and seven stateful pre-emptors, and every one of them
|
||||
// answers "is this mine?" alone. None can answer "is this more mine than
|
||||
// yours?", because their scores are not comparable: stage 0 asserts 1.0 by
|
||||
// fiat, the classifier reports a cosine, the LLM router derives one from
|
||||
@@ -56,8 +56,9 @@ const (
|
||||
// BandStructural — the claimant read the whole sentence and produced a
|
||||
// complete route, every slot its intent requires filled. The LLM router at
|
||||
// full confidence, and a stateful claimant holding a pending question.
|
||||
// Below BandAnchored on purpose: the four stateful claimants pre-empt
|
||||
// unconditionally today, and that is the V-558 defect.
|
||||
// Below BandAnchored on purpose: the stateful claimants pre-empt
|
||||
// unconditionally today, and that is the V-558 defect. There are seven of
|
||||
// them and preRouteLadder in cmd/mavend/decisiontrace.go is the roster.
|
||||
BandStructural
|
||||
|
||||
// BandAnchored — a literal pattern anchored in the utterance matched, and
|
||||
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
"encoding/base64"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"log"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
@@ -329,7 +330,15 @@ func Load(path string) (*Config, error) {
|
||||
// Expand ${VAR} or $VAR patterns from environment variables. This lets
|
||||
// secrets live in env (docker-compose env_file) rather than the config
|
||||
// file committed to git.
|
||||
expanded := os.ExpandEnv(string(b))
|
||||
expanded, missing := expandEnv(string(b))
|
||||
if len(missing) > 0 {
|
||||
// An unset variable expands to "", which every block reads as "not
|
||||
// configured" and none of them complains about. That is the intended
|
||||
// behaviour and it stays: CI parses this same file with no secrets
|
||||
// present. What was missing is the line telling the operator which
|
||||
// capability he just turned off by forgetting an env file.
|
||||
log.Printf("config: %s references unset environment variables %v — those settings are empty, so whatever they configure is off", path, missing)
|
||||
}
|
||||
var c Config
|
||||
if err := json.Unmarshal([]byte(expanded), &c); err != nil {
|
||||
return nil, fmt.Errorf("config: parse %s: %w", path, err)
|
||||
@@ -341,6 +350,23 @@ func Load(path string) (*Config, error) {
|
||||
return &c, nil
|
||||
}
|
||||
|
||||
// expandEnv is os.ExpandEnv plus the names it could not resolve, each reported
|
||||
// once and in the order the file mentions them. A variable set to the empty
|
||||
// string counts as set: the operator wrote it down, so he meant it.
|
||||
func expandEnv(s string) (string, []string) {
|
||||
var missing []string
|
||||
seen := map[string]bool{}
|
||||
out := os.Expand(s, func(name string) string {
|
||||
v, ok := os.LookupEnv(name)
|
||||
if !ok && !seen[name] {
|
||||
seen[name] = true
|
||||
missing = append(missing, name)
|
||||
}
|
||||
return v
|
||||
})
|
||||
return out, missing
|
||||
}
|
||||
|
||||
func (c *Config) applyDefaults() {
|
||||
if c.IntakeJournal == 0 {
|
||||
c.IntakeJournal = DefaultIntakeJournal
|
||||
|
||||
@@ -19,6 +19,13 @@ type Ring struct {
|
||||
func NewRing() *Ring { return &Ring{} }
|
||||
|
||||
// Push adds one finished record and drops the oldest past the bound.
|
||||
//
|
||||
// The dropped pointers are cleared before the reslice. Resliceing alone moves
|
||||
// the window forward and leaves the evicted records addressable from the
|
||||
// backing array, so up to ringSize turns he had already aged out stayed in
|
||||
// memory until the next append reallocated. That is a leak anywhere and it is
|
||||
// the wrong one here, because the reason this store is memory-only is that his
|
||||
// words should not outlive the diagnosis.
|
||||
func (r *Ring) Push(rec *Record) {
|
||||
if r == nil || rec == nil {
|
||||
return
|
||||
@@ -26,8 +33,11 @@ func (r *Ring) Push(rec *Record) {
|
||||
r.mu.Lock()
|
||||
defer r.mu.Unlock()
|
||||
r.recs = append(r.recs, rec)
|
||||
if len(r.recs) > ringSize {
|
||||
r.recs = r.recs[len(r.recs)-ringSize:]
|
||||
if drop := len(r.recs) - ringSize; drop > 0 {
|
||||
for i := 0; i < drop; i++ {
|
||||
r.recs[i] = nil
|
||||
}
|
||||
r.recs = r.recs[drop:]
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package email
|
||||
|
||||
import "strings"
|
||||
|
||||
// windows-1251 (and its ASCII-compatible low half) is decoded here rather than
|
||||
// pulled in from x/text.
|
||||
//
|
||||
@@ -35,14 +37,19 @@ var cp1251High = [128]rune{
|
||||
|
||||
// decodeCP1251 maps each byte through the table. Every byte has a defined
|
||||
// meaning in this charset, so decoding cannot fail.
|
||||
//
|
||||
// It writes into a Builder rather than collecting runes: a []rune of the whole
|
||||
// body is four bytes a character and was then copied again into the string, so
|
||||
// a 1 MiB cp1251 mail allocated about 6 MiB to produce roughly 2.
|
||||
func decodeCP1251(b []byte) string {
|
||||
out := make([]rune, 0, len(b))
|
||||
var out strings.Builder
|
||||
out.Grow(len(b))
|
||||
for _, c := range b {
|
||||
if c < 0x80 {
|
||||
out = append(out, rune(c))
|
||||
out.WriteByte(c)
|
||||
continue
|
||||
}
|
||||
out = append(out, cp1251High[c-0x80])
|
||||
out.WriteRune(cp1251High[c-0x80])
|
||||
}
|
||||
return string(out)
|
||||
return out.String()
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
package email
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
"io"
|
||||
@@ -57,7 +58,10 @@ type Message struct {
|
||||
// through, because a subject line alone is often the whole task ("Счёт за
|
||||
// интернет"). Only a message whose headers cannot be read at all is an error.
|
||||
func ParseMessage(uid uint32, raw []byte) (Message, error) {
|
||||
m, err := mail.ReadMessage(strings.NewReader(string(raw)))
|
||||
// bytes.NewReader, not strings.NewReader(string(raw)): the conversion copied
|
||||
// the whole message, and MaxMessageBytes lets that be 2 MiB per mail on a box
|
||||
// already holding the resident model.
|
||||
m, err := mail.ReadMessage(bytes.NewReader(raw))
|
||||
if err != nil {
|
||||
return Message{}, fmt.Errorf("email: parse message: %w", err)
|
||||
}
|
||||
@@ -82,6 +86,23 @@ func ParseMessage(uid uint32, raw []byte) (Message, error) {
|
||||
// wholesale — an attachment is a file, not a sentence, and reading one would
|
||||
// mean parsing arbitrary formats from the network.
|
||||
func plaintextBody(contentType, encoding string, body io.Reader) (string, error) {
|
||||
return plaintextBodyAt(contentType, encoding, body, 0)
|
||||
}
|
||||
|
||||
// MaxMIMEDepth — how deep the MIME tree is walked.
|
||||
//
|
||||
// The nesting comes off the wire, so the recursion depth is the sender's to
|
||||
// pick: a boundary line is a few bytes, and one message inside MaxMessageBytes
|
||||
// can declare tens of thousands of multipart levels. Real mail is three deep
|
||||
// (mixed, then alternative, then related), so a message past this is malformed
|
||||
// or hostile and truncating the walk costs a body nobody was going to read.
|
||||
const MaxMIMEDepth = 12
|
||||
|
||||
// plaintextBodyAt is plaintextBody carrying the current nesting depth.
|
||||
func plaintextBodyAt(contentType, encoding string, body io.Reader, depth int) (string, error) {
|
||||
if depth > MaxMIMEDepth {
|
||||
return "", nil
|
||||
}
|
||||
mediaType, params, err := mime.ParseMediaType(contentType)
|
||||
if contentType == "" || err != nil {
|
||||
// No Content-Type at all is legal and means text/plain; a broken one is
|
||||
@@ -94,7 +115,7 @@ func plaintextBody(contentType, encoding string, body io.Reader) (string, error)
|
||||
if boundary == "" {
|
||||
return "", fmt.Errorf("email: multipart without boundary")
|
||||
}
|
||||
plain, html, err := multipartText(multipart.NewReader(body, boundary))
|
||||
plain, html, err := multipartText(multipart.NewReader(body, boundary), depth+1)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
@@ -124,7 +145,7 @@ func plaintextBody(contentType, encoding string, body io.Reader) (string, error)
|
||||
// contribute either kind. Folding a nested level's answer into one string put
|
||||
// HTML-derived text in the plain bucket, and a real text/plain sibling later in
|
||||
// the message was then thrown away by the "plain is already set" guard.
|
||||
func multipartText(mr *multipart.Reader) (plain, html string, err error) {
|
||||
func multipartText(mr *multipart.Reader, depth int) (plain, html string, err error) {
|
||||
for {
|
||||
part, err := mr.NextPart()
|
||||
if err == io.EOF {
|
||||
@@ -143,8 +164,8 @@ func multipartText(mr *multipart.Reader) (plain, html string, err error) {
|
||||
switch {
|
||||
case strings.HasPrefix(mediaType, "multipart/"):
|
||||
var np, nh string
|
||||
if b := params["boundary"]; b != "" {
|
||||
np, nh, _ = multipartText(multipart.NewReader(part, b))
|
||||
if b := params["boundary"]; b != "" && depth <= MaxMIMEDepth {
|
||||
np, nh, _ = multipartText(multipart.NewReader(part, b), depth+1)
|
||||
}
|
||||
part.Close()
|
||||
if plain == "" {
|
||||
@@ -154,7 +175,7 @@ func multipartText(mr *multipart.Reader) (plain, html string, err error) {
|
||||
html = nh
|
||||
}
|
||||
default:
|
||||
text, terr := plaintextBody(ct, part.Header.Get("Content-Transfer-Encoding"), part)
|
||||
text, terr := plaintextBodyAt(ct, part.Header.Get("Content-Transfer-Encoding"), part, depth)
|
||||
part.Close()
|
||||
if terr != nil || strings.TrimSpace(text) == "" {
|
||||
continue
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package email
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
@@ -143,6 +144,24 @@ func TestParseTruncatesLongBody(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Nesting depth comes off the wire, so a hostile message must not get to pick
|
||||
// the recursion depth. The walk stops and the headers still come through.
|
||||
func TestParseMessageBoundsMIMEDepth(t *testing.T) {
|
||||
var b strings.Builder
|
||||
b.WriteString("Subject: deep\r\nMIME-Version: 1.0\r\n")
|
||||
for i := 0; i < MaxMIMEDepth+20; i++ {
|
||||
fmt.Fprintf(&b, "Content-Type: multipart/mixed; boundary=\"b%d\"\r\n\r\n--b%d\r\n", i, i)
|
||||
}
|
||||
b.WriteString("Content-Type: text/plain\r\n\r\nглубоко\r\n")
|
||||
msg, err := ParseMessage(7, []byte(b.String()))
|
||||
if err != nil {
|
||||
t.Fatalf("ParseMessage: %v", err)
|
||||
}
|
||||
if msg.Subject != "deep" {
|
||||
t.Errorf("Subject = %q, want the headers to survive", msg.Subject)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCollapseSqueezesBlankLines(t *testing.T) {
|
||||
got := collapse(" a b \r\n\r\n\r\n\r\n c \r\n")
|
||||
if got != "a b\n\nc" {
|
||||
|
||||
@@ -131,9 +131,15 @@ const (
|
||||
codeInternal = "internal"
|
||||
)
|
||||
|
||||
// codeOf maps a server-side sentinel to its wire code. Anything not matched
|
||||
// is codeInternal — we never leak internal Go error text to a module; it
|
||||
// gets a generic "internal" and the daemon logs the real error server-side.
|
||||
// codeOf maps a server-side sentinel to its wire code. Anything not matched is
|
||||
// codeInternal.
|
||||
//
|
||||
// This used to claim the text of an unmatched error stays server-side. It does
|
||||
// not: rpcErr below ships err.Error() for codeInternal and codeBadParams,
|
||||
// deliberately, because on those two codes the text is the whole diagnostic and
|
||||
// a module has no other way to see it. Worth knowing before putting a secret in
|
||||
// an error string, and worth knowing twice on a tcp seam, where that string
|
||||
// leaves the box.
|
||||
func codeOf(err error) string {
|
||||
switch {
|
||||
case err == nil:
|
||||
|
||||
@@ -62,10 +62,10 @@ func mustLoad() lexiconFile {
|
||||
}
|
||||
for _, name := range []string{
|
||||
"interrogatives", "capture_verbs", "narrative_requests", "cardinals", "ordinals",
|
||||
"day_offsets", "weekdays", "months_genitive", "hours_spoken",
|
||||
"day_offsets", "weekdays", "weekdays_english", "months_genitive", "hours_spoken",
|
||||
"not_place_after_v", "parts_of_day", "reminder_verbs", "half_hour",
|
||||
"filler_particles", "task_done_words", "task_drop_words",
|
||||
"confirm_yes", "confirm_no",
|
||||
"confirm_yes", "confirm_no", "hour_units", "minute_units",
|
||||
} {
|
||||
s, ok := f.Sets[name]
|
||||
if !ok || (len(s.Words) == 0 && len(s.Values) == 0) {
|
||||
@@ -139,7 +139,40 @@ func TaskDropWords() []string { return words("task_drop_words") }
|
||||
// making the utterance a request of its own. A caller strips these (along with
|
||||
// the numbers and the other closed time sets) to see whether an utterance
|
||||
// carries any content beside the value it was asked for. See the set's note.
|
||||
func SlotValueFrame() []string { return words("slot_value_frame") }
|
||||
// The hour and the minute nouns are part of the frame and are kept in their own
|
||||
// sets, so there is one copy of each closed class rather than a copy per caller.
|
||||
func SlotValueFrame() []string {
|
||||
out := words("slot_value_frame")
|
||||
out = append(out, HourUnits()...)
|
||||
out = append(out, MinuteUnits()...)
|
||||
return out
|
||||
}
|
||||
|
||||
// HourUnits returns every form of the hour noun, and MinuteUnits every form of
|
||||
// the minute noun. One home for each, because four router sets used to list the
|
||||
// hour and all four stopped at "часу" (V-609). A caller folding time words into
|
||||
// one set reads these; a caller asking about a single word reads IsHourUnit or
|
||||
// IsMinuteUnit.
|
||||
func HourUnits() []string { return words("hour_units") }
|
||||
|
||||
// MinuteUnits — see HourUnits.
|
||||
func MinuteUnits() []string { return words("minute_units") }
|
||||
|
||||
// IsHourUnit reports whether a word is the hour noun in any form.
|
||||
func IsHourUnit(word string) bool { return inSet("hour_units", word) }
|
||||
|
||||
// IsMinuteUnit reports whether a word is the minute noun in any form.
|
||||
func IsMinuteUnit(word string) bool { return inSet("minute_units", word) }
|
||||
|
||||
func inSet(set, word string) bool {
|
||||
w := norm(word)
|
||||
for _, s := range ru.Sets[set].Words {
|
||||
if w == s {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// DialogueCancel returns the ways he calls off the request Maven is assembling.
|
||||
// Distinct from TaskDropWords, which abandons an item that already exists.
|
||||
@@ -283,6 +316,20 @@ func DayOffsetIn(text string) (int, bool) {
|
||||
// Go's time.Weekday. An index off the end returns "".
|
||||
func Weekday(i int) string { return at("weekdays", i) }
|
||||
|
||||
// Weekdays returns the seven Russian names in one slice, Sunday first, for a
|
||||
// caller matching a token against all of them rather than rendering one. Only
|
||||
// the nominative is here: every other case lemmatises to it, so an oblique form
|
||||
// is morph's question and not a second list (V-581).
|
||||
func Weekdays() []string { return words("weekdays") }
|
||||
|
||||
// WeekdayEnglish reports the Go time.Weekday index an English weekday names,
|
||||
// singular or plural. English needs the list that Russian does not, because the
|
||||
// vendored dictionary is Russian and leaves "mondays" as it found it.
|
||||
func WeekdayEnglish(word string) (int, bool) {
|
||||
n, ok := ru.Sets["weekdays_english"].Values[norm(word)]
|
||||
return n, ok
|
||||
}
|
||||
|
||||
// MonthGenitive returns the month name a date takes — "10 июля", not "июль".
|
||||
// The set is 1-indexed, so MonthGenitive(int(t.Month())) is the whole call.
|
||||
func MonthGenitive(m int) string { return at("months_genitive", m) }
|
||||
|
||||
@@ -56,13 +56,13 @@
|
||||
}
|
||||
},
|
||||
"cardinals": {
|
||||
"note": "Number words as spoken, with the gender variants Russian requires (один/одна/одно and два/две agree with the noun that follows) and the oblique forms, because a spoken time declines: \"в семь\", \"к семи\", \"около семи\" are three forms of one hour (Vikunja #530). Values are the number itself. Twenties and up are compounds and are read as their parts, so only the round members are listed.",
|
||||
"note": "Number words as spoken, with the gender variants Russian requires (один/одна/одно and два/две agree with the noun that follows) and the oblique forms, because a spoken time declines: \"в семь\", \"к семи\", \"около семи\" are three forms of one hour (Vikunja #530). Values are the number itself. Twenties and up are compounds and are read as their parts, so only the round members are listed. From five up one oblique form serves the genitive, dative and prepositional, so \"пяти\" is the whole set; one to four decline separately and carry the dative and instrumental of their own, because \"к двум часам\" and \"к трём\" are hours he says (V-581).",
|
||||
"values": {
|
||||
"ноль": 0, "нуль": 0, "zero": 0,
|
||||
"один": 1, "одна": 1, "одно": 1, "одного": 1, "одной": 1, "одну": 1, "one": 1,
|
||||
"два": 2, "две": 2, "двух": 2, "two": 2,
|
||||
"три": 3, "трёх": 3, "трех": 3, "three": 3,
|
||||
"четыре": 4, "четырёх": 4, "четырех": 4, "four": 4,
|
||||
"один": 1, "одна": 1, "одно": 1, "одного": 1, "одной": 1, "одну": 1, "одному": 1, "одним": 1, "one": 1,
|
||||
"два": 2, "две": 2, "двух": 2, "двум": 2, "двумя": 2, "two": 2,
|
||||
"три": 3, "трёх": 3, "трех": 3, "трём": 3, "трем": 3, "тремя": 3, "three": 3,
|
||||
"четыре": 4, "четырёх": 4, "четырех": 4, "четырём": 4, "четырем": 4, "четырьмя": 4, "four": 4,
|
||||
"пять": 5, "пяти": 5, "five": 5,
|
||||
"шесть": 6, "шести": 6, "six": 6,
|
||||
"семь": 7, "семи": 7, "seven": 7,
|
||||
@@ -111,6 +111,18 @@
|
||||
"четверг", "пятница", "суббота"
|
||||
]
|
||||
},
|
||||
"weekdays_english": {
|
||||
"note": "The English weekday names with their Go time.Weekday index, plus the plural a habit is spoken in (\"on mondays\"). English is listed as words where Russian is not, because the vendored dictionary is Russian: it lemmatises \"пятницу\" to \"пятница\" on its own and leaves \"mondays\" alone (V-581). So the Russian side of a weekday match is grammar and the English side is data.",
|
||||
"values": {
|
||||
"sunday": 0, "sundays": 0,
|
||||
"monday": 1, "mondays": 1,
|
||||
"tuesday": 2, "tuesdays": 2,
|
||||
"wednesday": 3, "wednesdays": 3,
|
||||
"thursday": 4, "thursdays": 4,
|
||||
"friday": 5, "fridays": 5,
|
||||
"saturday": 6, "saturdays": 6
|
||||
}
|
||||
},
|
||||
"months_genitive": {
|
||||
"note": "The form a date takes: \"10 июля\", not \"июль\". 1-indexed, so slot 0 is empty and month numbers need no arithmetic.",
|
||||
"words": [
|
||||
@@ -198,6 +210,20 @@
|
||||
"передумал", "передумала", "неактуально"
|
||||
]
|
||||
},
|
||||
"hour_units": {
|
||||
"note": "Every form of the hour noun, Russian and English (V-609). One home for a closed class that four router sets used to list separately, and all four stopped at \"часу\": \"напомни к двум часам\" lost its hour and the reminder was left asking \"Когда?\". Russian declines, so the dative plural is as ordinary a way to say an hour as the accusative singular. A caller that folds time words into one set reads HourUnits; a caller asking about one word reads IsHourUnit.",
|
||||
"words": [
|
||||
"час", "часа", "часов", "часу", "часам", "часами", "часах",
|
||||
"hour", "hours"
|
||||
]
|
||||
},
|
||||
"minute_units": {
|
||||
"note": "Every form of the minute noun, Russian and English (V-609). Same class as hour_units one noun over, and it had the same gap: the dative plural \"минутам\" was missing everywhere \"минут\" and \"минуты\" were present.",
|
||||
"words": [
|
||||
"минута", "минуты", "минуту", "минут", "минуте", "минутам", "минутами", "минутах",
|
||||
"minute", "minutes"
|
||||
]
|
||||
},
|
||||
"slot_value_frame": {
|
||||
"note": "The words that can stand around a bare slot value without making the utterance a request of its own (Vikunja #560). Prepositions, hedges and the nouns a spoken time is built from: strip these, the numbers, the interrogatives, the filler particles and the other time sets, and whatever is left is the utterance's OWN content. \"а что если в 11:00\" leaves nothing and is an answer; \"какая сейчас погода в Риме\" leaves \"погода\" and \"Риме\" and is not. Closed because each part of it is closed — Russian has a fixed list of prepositions, and a clock is built from a fixed list of nouns. It is not a stopword list: a word goes in only if it can never be the thing he is asking about.",
|
||||
"words": [
|
||||
@@ -205,10 +231,10 @@
|
||||
"at", "on", "in", "by", "to", "till", "until", "after", "before", "about", "for",
|
||||
"нет", "не", "да", "ага", "угу", "ой", "ох", "тогда", "лучше", "может", "можно", "наверное", "наверно", "пожалуй", "точнее", "скорее", "если", "пусть", "прости", "извини", "слушай", "значит", "как-то", "типа", "вообще-то",
|
||||
"no", "yes", "yeah", "ok", "okay", "sorry", "maybe", "actually", "rather", "then", "well",
|
||||
"час", "часа", "часов", "часу", "часам", "минут", "минута", "минуты", "минуту", "минутах", "полдень", "полночь", "полдня",
|
||||
"полдень", "полночь", "полдня",
|
||||
"утра", "утро", "утру", "дня", "день", "днями", "вечера", "вечер", "вечеру", "ночи", "ночь", "ночью",
|
||||
"сейчас", "теперь", "сегодняшний", "ближайший", "ближайшее",
|
||||
"hour", "hours", "minute", "minutes", "noon", "midnight", "am", "pm", "oclock", "now"
|
||||
"noon", "midnight", "am", "pm", "oclock", "now"
|
||||
]
|
||||
},
|
||||
"dialogue_cancel": {
|
||||
|
||||
@@ -36,6 +36,46 @@ func TestClosedSetsAreComplete(t *testing.T) {
|
||||
if _, ok := Cardinal("бэкап"); ok {
|
||||
t.Error("Cardinal must not answer for a word that is not a number")
|
||||
}
|
||||
|
||||
// A spoken hour declines, and one to four decline further than the rest:
|
||||
// "к двум часам" and "к трём" are hours, and only the dative says so (V-581).
|
||||
for _, tc := range []struct {
|
||||
word string
|
||||
want int
|
||||
}{
|
||||
{"одному", 1}, {"двум", 2}, {"двумя", 2}, {"трём", 3}, {"трем", 3},
|
||||
{"четырём", 4}, {"четырем", 4}, {"пяти", 5}, {"семи", 7},
|
||||
} {
|
||||
if got, ok := Cardinal(tc.word); !ok || got != tc.want {
|
||||
t.Errorf("Cardinal(%q) = %d, %v; want %d, true", tc.word, got, ok, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestWeekdaysAreOneList — the second copy of a closed class is the bug (V-581).
|
||||
// Weekdays lived in four files outside this one, so the list is handed out whole
|
||||
// and the English forms, which the Russian dictionary cannot lemmatise, are here.
|
||||
func TestWeekdaysAreOneList(t *testing.T) {
|
||||
days := Weekdays()
|
||||
if len(days) != 7 || days[0] != "воскресенье" || days[1] != "понедельник" {
|
||||
t.Fatalf("Weekdays() = %v; want the seven, Sunday first", days)
|
||||
}
|
||||
for i, name := range days {
|
||||
if Weekday(i) != name {
|
||||
t.Errorf("Weekdays()[%d] = %q, but Weekday(%d) = %q", i, name, i, Weekday(i))
|
||||
}
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
word string
|
||||
want int
|
||||
}{{"sunday", 0}, {"monday", 1}, {"mondays", 1}, {"Friday", 5}, {"saturdays", 6}} {
|
||||
if got, ok := WeekdayEnglish(tc.word); !ok || got != tc.want {
|
||||
t.Errorf("WeekdayEnglish(%q) = %d, %v; want %d, true", tc.word, got, ok, tc.want)
|
||||
}
|
||||
}
|
||||
if _, ok := WeekdayEnglish("понедельник"); ok {
|
||||
t.Error("WeekdayEnglish answered for a Russian word; that side is morph's")
|
||||
}
|
||||
}
|
||||
|
||||
// TestDayOffsetHasNoOrderingTrap — the defect a lookup removes. The callers this
|
||||
|
||||
+33
-7
@@ -5,6 +5,7 @@ import (
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
)
|
||||
@@ -46,6 +47,7 @@ type Pair struct {
|
||||
interval time.Duration
|
||||
http *http.Client
|
||||
stop chan struct{}
|
||||
stopOnce sync.Once
|
||||
}
|
||||
|
||||
// ErrRemoteUnavailable — the workstation model was required and is not
|
||||
@@ -107,13 +109,11 @@ func (p *Pair) Start(ctx context.Context) {
|
||||
}()
|
||||
}
|
||||
|
||||
// Stop ends the prober. Idempotent.
|
||||
// Stop ends the prober. Idempotent, and safe from two goroutines at once. The
|
||||
// check-then-close it replaced let both callers see an open channel and the
|
||||
// second close panicked, which turned a shutdown race into a crash.
|
||||
func (p *Pair) Stop() {
|
||||
select {
|
||||
case <-p.stop:
|
||||
default:
|
||||
close(p.stop)
|
||||
}
|
||||
p.stopOnce.Do(func() { close(p.stop) })
|
||||
}
|
||||
|
||||
// Available reports whether the workstation will take work right now. It reads
|
||||
@@ -170,7 +170,9 @@ func (p *Pair) Complete(ctx context.Context, r Req) (string, error) {
|
||||
}
|
||||
why := "workstation down"
|
||||
if p.Available() {
|
||||
out, err := p.remote.Complete(ctx, r)
|
||||
rctx, cancel := remoteBudget(ctx)
|
||||
out, err := p.remote.Complete(rctx, r)
|
||||
cancel()
|
||||
if err == nil {
|
||||
log.Print("llm: served by the workstation model")
|
||||
return out, nil
|
||||
@@ -184,6 +186,30 @@ func (p *Pair) Complete(ctx context.Context, r Req) (string, error) {
|
||||
return p.floor.Complete(ctx, r)
|
||||
}
|
||||
|
||||
// remoteBudget bounds the workstation attempt so the floor still has time to
|
||||
// answer. A turn carrying a deadline used to hand the whole of it to the
|
||||
// remote, so a workstation that accepted the connection and then hung ate the
|
||||
// budget and the fallback ran on an already-expired context: the floor
|
||||
// returned the deadline error and the turn broke on the workstation being
|
||||
// slow, which docs/offload.md says must never happen. Half is the split
|
||||
// because both halves have to be able to finish, and there is no reason to
|
||||
// prefer either one when the remote is the part that failed.
|
||||
//
|
||||
// A context with no deadline is left alone. The remote client's own timeout
|
||||
// (workstation.timeout, 90s by default) bounds it there, and shortening that
|
||||
// silently would change the configured budget.
|
||||
func remoteBudget(ctx context.Context) (context.Context, context.CancelFunc) {
|
||||
dl, ok := ctx.Deadline()
|
||||
if !ok {
|
||||
return ctx, func() {}
|
||||
}
|
||||
left := time.Until(dl)
|
||||
if left <= 0 {
|
||||
return ctx, func() {}
|
||||
}
|
||||
return context.WithTimeout(ctx, left/2)
|
||||
}
|
||||
|
||||
// CompleteRemote runs r on the workstation or refuses. It never falls back,
|
||||
// because for a world question the resident 1.7B does not answer worse, it
|
||||
// invents. Callers turn ErrRemoteUnavailable into a named gap.
|
||||
|
||||
@@ -154,6 +154,48 @@ func TestRemoteErrorMidRequestFallsBack(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A workstation that accepts the connection and then hangs must not spend the
|
||||
// whole turn budget. It used to: the remote got the caller's context unchanged,
|
||||
// so the fallback ran on an expired one and the floor returned the deadline
|
||||
// error instead of an answer. The turn broke on the workstation being slow,
|
||||
// which is the one outcome docs/offload.md rules out.
|
||||
func TestHangingRemoteLeavesTheFloorABudget(t *testing.T) {
|
||||
var floorHits atomic.Int64
|
||||
// released, not r.Context().Done(): httptest.Server.Close waits for the
|
||||
// handler, and a handler that only watches the request context can outlive
|
||||
// the test when the client hangs up without the server noticing.
|
||||
released := make(chan struct{})
|
||||
hang := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
select {
|
||||
case <-released:
|
||||
case <-r.Context().Done():
|
||||
}
|
||||
}))
|
||||
defer hang.Close()
|
||||
defer close(released)
|
||||
floor := completionServer(t, "floor", &floorHits)
|
||||
up := &atomic.Bool{}
|
||||
up.Store(true)
|
||||
health := healthServer(t, up)
|
||||
|
||||
p := NewPair(New(hang.URL, time.Minute), New(floor.URL, time.Minute), health.URL, time.Hour)
|
||||
p.Start(context.Background())
|
||||
defer p.Stop()
|
||||
if !waitFor(t, p.Available) {
|
||||
t.Fatal("prober never saw the remote come up")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 400*time.Millisecond)
|
||||
defer cancel()
|
||||
out, err := p.Complete(ctx, Req{User: "привет"})
|
||||
if err != nil {
|
||||
t.Fatalf("complete: %v", err)
|
||||
}
|
||||
if out != "floor" || floorHits.Load() != 1 {
|
||||
t.Fatalf("out = %q, floor hits = %d", out, floorHits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// The naming half of the degradation rule. A world question must not be handed
|
||||
// to the resident model, because it answers by inventing.
|
||||
func TestCompleteRemoteNamesTheGap(t *testing.T) {
|
||||
|
||||
+22
-4
@@ -265,6 +265,16 @@ func (s *Store) Put(kind Kind, mime, source string, data []byte) (Blob, error) {
|
||||
}
|
||||
}
|
||||
|
||||
// Every return past the reservation has to give it back, so the defer owns
|
||||
// that rather than each error path: a path that forgot over-counted the
|
||||
// store until the next Open re-walked the directory.
|
||||
stored := false
|
||||
defer func() {
|
||||
if fresh && !stored {
|
||||
s.release(b.Size)
|
||||
}
|
||||
}()
|
||||
|
||||
// The sidecar goes first. Written second, a full disk or a crash between
|
||||
// the two left the bytes on disk with no sidecar, and List only sees
|
||||
// sidecars, so Prune could never collect them: Put returned an error and an
|
||||
@@ -274,11 +284,9 @@ func (s *Store) Put(kind Kind, mime, source string, data []byte) (Blob, error) {
|
||||
}
|
||||
if err := writeFile(blobPath, data); err != nil {
|
||||
_ = os.Remove(metaPath)
|
||||
if fresh {
|
||||
s.release(b.Size)
|
||||
}
|
||||
return Blob{}, err
|
||||
}
|
||||
stored = true
|
||||
return b, nil
|
||||
}
|
||||
|
||||
@@ -331,12 +339,22 @@ func (s *Store) PutFile(kind Kind, mime, source, src string) (Blob, error) {
|
||||
return Blob{}, err
|
||||
}
|
||||
}
|
||||
// Same reasoning as Put: the reservation is released by one defer, not by
|
||||
// whichever error path remembered to.
|
||||
stored := false
|
||||
defer func() {
|
||||
if fresh && !stored {
|
||||
s.release(b.Size)
|
||||
}
|
||||
}()
|
||||
|
||||
if err := writeMeta(metaPath, b); err != nil {
|
||||
return Blob{}, err
|
||||
}
|
||||
if !fresh {
|
||||
// Same bytes already here. Drop the spool copy.
|
||||
_ = os.Remove(src)
|
||||
stored = true
|
||||
return b, nil
|
||||
}
|
||||
if err := os.Chmod(src, filePerm); err != nil {
|
||||
@@ -344,9 +362,9 @@ func (s *Store) PutFile(kind Kind, mime, source, src string) (Blob, error) {
|
||||
}
|
||||
if err := os.Rename(src, blobPath); err != nil {
|
||||
_ = os.Remove(metaPath)
|
||||
s.release(b.Size)
|
||||
return Blob{}, fmt.Errorf("media: move spool: %w", err)
|
||||
}
|
||||
stored = true
|
||||
return b, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -287,6 +287,73 @@ func TestPutLeavesNothingWhenTheBytesCannotBeWritten(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// An over-counted store answers ErrStoreFull while the disk has room, and only
|
||||
// the next Open corrects it. So every failed write has to give its reservation
|
||||
// back, not just the one that remembered to.
|
||||
func TestPutReleasesTheBudgetWhenTheSidecarCannotBeWritten(t *testing.T) {
|
||||
s := testStore(t)
|
||||
data := []byte("no sidecar for this")
|
||||
blockSidecar(t, s, KindImage, data)
|
||||
if _, err := s.Put(KindImage, "image/png", "web:upload", data); err == nil {
|
||||
t.Fatal("put must fail")
|
||||
}
|
||||
if s.Total() != 0 {
|
||||
t.Errorf("total = %d, want the failed put not counted", s.Total())
|
||||
}
|
||||
}
|
||||
|
||||
func TestPutFileReleasesTheBudgetWhenTheSidecarCannotBeWritten(t *testing.T) {
|
||||
s := testStore(t)
|
||||
data := []byte("no sidecar for this either")
|
||||
blockSidecar(t, s, KindAudio, data)
|
||||
src := filepath.Join(t.TempDir(), "capture.wav")
|
||||
if err := os.WriteFile(src, data, 0o600); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := s.PutFile(KindAudio, "audio/wav", "meeting", src); err == nil {
|
||||
t.Fatal("put file must fail")
|
||||
}
|
||||
if s.Total() != 0 {
|
||||
t.Errorf("total = %d, want the failed put not counted", s.Total())
|
||||
}
|
||||
}
|
||||
|
||||
// The chmod arm is PutFile's alone: Put never touches a spool file.
|
||||
func TestPutFileReleasesTheBudgetWhenTheSpoolCannotBeChmodded(t *testing.T) {
|
||||
if os.Geteuid() == 0 {
|
||||
t.Skip("root can chmod a file it does not own")
|
||||
}
|
||||
// A symlink to a file owned by somebody else. Stat and the hash follow it
|
||||
// and succeed; chmod follows it too and is refused.
|
||||
src := filepath.Join(t.TempDir(), "capture.wav")
|
||||
if err := os.Symlink("/etc/hosts", src); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
info, err := os.Stat(src)
|
||||
if err != nil || info.Size() == 0 {
|
||||
t.Skip("no readable /etc/hosts to point at")
|
||||
}
|
||||
s := testStore(t)
|
||||
if _, err := s.PutFile(KindAudio, "audio/wav", "meeting", src); err == nil {
|
||||
t.Fatal("put file must fail")
|
||||
}
|
||||
if s.Total() != 0 {
|
||||
t.Errorf("total = %d, want the failed put not counted", s.Total())
|
||||
}
|
||||
}
|
||||
|
||||
// blockSidecar puts a directory where the sidecar for data has to go, so
|
||||
// writeMeta fails while the blob path is still free.
|
||||
func blockSidecar(t *testing.T, s *Store, kind Kind, data []byte) {
|
||||
t.Helper()
|
||||
sum := sha256.Sum256(data)
|
||||
id := hex.EncodeToString(sum[:])
|
||||
bucket := filepath.Join(s.dir, string(kind), id[:2])
|
||||
if err := os.MkdirAll(filepath.Join(bucket, id+".json"), 0o700); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
// The per-blob cap bounds one call and nothing bounded their sum. 64 MiB per
|
||||
// call times unlimited calls inside a seven-day window fills the disk mavend's
|
||||
// database lives on.
|
||||
|
||||
+69
-11
@@ -31,6 +31,7 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"golang.org/x/sys/unix"
|
||||
@@ -154,31 +155,78 @@ func clientHandshake(c net.Conn, token string) error {
|
||||
// Listener wraps a net.Listener so Accept performs the token check for a tcp
|
||||
// seam. A connection that fails the check is closed and never surfaces, so
|
||||
// the protocol above this layer only ever sees authorized peers.
|
||||
//
|
||||
// A unix seam takes none of that machinery: Accept delegates straight to the
|
||||
// wrapped listener, which is what it did before the token existed.
|
||||
type Listener struct {
|
||||
net.Listener
|
||||
addr Addr
|
||||
|
||||
start sync.Once
|
||||
closeOnce sync.Once
|
||||
conns chan net.Conn
|
||||
errc chan error // buffered 1, re-armed so every Accept sees the error
|
||||
done chan struct{}
|
||||
}
|
||||
|
||||
// Accept returns the next authorized connection. Unauthorized peers are
|
||||
// dropped and Accept keeps waiting: a bad token is a rejected stranger, not a
|
||||
// reason to stop serving.
|
||||
//
|
||||
// Each tcp handshake runs in its own goroutine rather than inline here. A peer
|
||||
// that connects and then says nothing holds its greeting open for
|
||||
// handshakeTimeout, and inline that peer stalls every other connection for
|
||||
// five seconds — one silent stranger was enough to freeze the seam.
|
||||
func (l *Listener) Accept() (net.Conn, error) {
|
||||
if l.addr.IsUnix() {
|
||||
return l.Listener.Accept()
|
||||
}
|
||||
l.start.Do(func() { go l.acceptLoop() })
|
||||
select {
|
||||
case c := <-l.conns:
|
||||
return c, nil
|
||||
case err := <-l.errc:
|
||||
l.errc <- err
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
|
||||
// acceptLoop takes connections off the wrapped listener and greets each one
|
||||
// concurrently. It ends on the first listener error, which every later Accept
|
||||
// then reports.
|
||||
func (l *Listener) acceptLoop() {
|
||||
for {
|
||||
c, err := l.Listener.Accept()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
select {
|
||||
case l.errc <- err:
|
||||
case <-l.done:
|
||||
}
|
||||
return
|
||||
}
|
||||
if l.addr.IsUnix() {
|
||||
return c, nil
|
||||
}
|
||||
if err := serverHandshake(c, l.addr.Token); err != nil {
|
||||
_ = c.Close()
|
||||
continue
|
||||
}
|
||||
return c, nil
|
||||
go l.greet(c)
|
||||
}
|
||||
}
|
||||
|
||||
func (l *Listener) greet(c net.Conn) {
|
||||
if err := serverHandshake(c, l.addr.Token); err != nil {
|
||||
_ = c.Close()
|
||||
return
|
||||
}
|
||||
select {
|
||||
case l.conns <- c:
|
||||
case <-l.done:
|
||||
_ = c.Close()
|
||||
}
|
||||
}
|
||||
|
||||
// Close stops the listener and releases any connection still waiting to be
|
||||
// handed to Accept.
|
||||
func (l *Listener) Close() error {
|
||||
l.closeOnce.Do(func() { close(l.done) })
|
||||
return l.Listener.Close()
|
||||
}
|
||||
|
||||
// Addr reports the parsed seam address this listener was built from.
|
||||
func (l *Listener) SeamAddr() Addr { return l.addr }
|
||||
|
||||
@@ -236,7 +284,7 @@ func Listen(a Addr) (*Listener, error) {
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &Listener{Listener: ln, addr: a}, nil
|
||||
return wrap(ln, a), nil
|
||||
}
|
||||
if a.Token == "" {
|
||||
return nil, fmt.Errorf("netaddr: listen %s: tcp seam requires a token", a)
|
||||
@@ -245,7 +293,17 @@ func Listen(a Addr) (*Listener, error) {
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("netaddr: listen %s: %w", a, err)
|
||||
}
|
||||
return &Listener{Listener: ln, addr: a}, nil
|
||||
return wrap(ln, a), nil
|
||||
}
|
||||
|
||||
func wrap(ln net.Listener, a Addr) *Listener {
|
||||
return &Listener{
|
||||
Listener: ln,
|
||||
addr: a,
|
||||
conns: make(chan net.Conn),
|
||||
errc: make(chan error, 1),
|
||||
done: make(chan struct{}),
|
||||
}
|
||||
}
|
||||
|
||||
func listenUnix(path string) (net.Listener, error) {
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"net"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A scheme-less address must stay unix. Every deploy in the tree writes a bare
|
||||
@@ -140,6 +141,41 @@ func TestTCPUngreetedPeerDoesNotKillTheListener(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A peer that connects and never speaks must not hold the seam. The greeting
|
||||
// it owes is bounded by handshakeTimeout, so serving it on the accept path
|
||||
// costs every later connection those five seconds.
|
||||
func TestTCPSilentPeerDoesNotStallTheSeam(t *testing.T) {
|
||||
ln, addr := listenLoopback(t, "s3cret")
|
||||
defer ln.Close()
|
||||
go echoOnce(ln)
|
||||
|
||||
mute, err := net.Dial("tcp", addr.Address)
|
||||
if err != nil {
|
||||
t.Fatalf("mute dial: %v", err)
|
||||
}
|
||||
defer mute.Close()
|
||||
|
||||
done := make(chan string, 1)
|
||||
go func() {
|
||||
c, err := Dial(addr)
|
||||
if err != nil {
|
||||
done <- "dial: " + err.Error()
|
||||
return
|
||||
}
|
||||
defer c.Close()
|
||||
done <- roundTrip(t, c, "still here")
|
||||
}()
|
||||
|
||||
select {
|
||||
case got := <-done:
|
||||
if got != "still here" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
case <-time.After(handshakeTimeout / 2):
|
||||
t.Fatal("a silent peer stalled the listener")
|
||||
}
|
||||
}
|
||||
|
||||
// A tcp seam with no token is a misconfiguration, and it must fail at bind
|
||||
// rather than serve the owner's turns to anyone who connects.
|
||||
func TestTCPListenRequiresToken(t *testing.T) {
|
||||
|
||||
@@ -33,6 +33,7 @@ import (
|
||||
"net/netip"
|
||||
"os"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -323,7 +324,7 @@ scan:
|
||||
go func(addr string, port int) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
if s.dial(ctx, net.JoinHostPort(addr, itoa(port)), s.cfg.Timeout) {
|
||||
if s.dial(ctx, net.JoinHostPort(addr, strconv.Itoa(port)), s.cfg.Timeout) {
|
||||
results <- result{addr: addr, ports: []int{port}}
|
||||
}
|
||||
}(addr, port)
|
||||
@@ -371,8 +372,6 @@ scan:
|
||||
return Result{Hosts: out, Truncated: truncated}, nil
|
||||
}
|
||||
|
||||
func itoa(n int) string { return fmt.Sprintf("%d", n) }
|
||||
|
||||
func dialTCP(ctx context.Context, addr string, timeout time.Duration) bool {
|
||||
d := net.Dialer{Timeout: timeout}
|
||||
ctx, cancel := context.WithTimeout(ctx, timeout)
|
||||
|
||||
@@ -17,6 +17,8 @@ import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/lexicon"
|
||||
)
|
||||
|
||||
// Facts — the optional, deployment-specific half of the block. All fields may
|
||||
@@ -34,12 +36,10 @@ type Facts struct {
|
||||
Tools bool // at least one shell act is on the allowlist
|
||||
}
|
||||
|
||||
var ruWeekdays = [...]string{"воскресенье", "понедельник", "вторник", "среда", "четверг", "пятница", "суббота"}
|
||||
|
||||
var ruMonths = [...]string{
|
||||
"января", "февраля", "марта", "апреля", "мая", "июня",
|
||||
"июля", "августа", "сентября", "октября", "ноября", "декабря",
|
||||
}
|
||||
// The weekday and month names are closed classes and live in internal/lexicon,
|
||||
// which indexes weekdays from Sunday the way time.Weekday does and months from
|
||||
// one. This file used to carry its own copies, making four copies of the twelve
|
||||
// months in the tree after cmd/mavend/ruwords.go gave up its own (Vikunja #525).
|
||||
|
||||
// Block renders the context block for one turn. Russian even in front of the
|
||||
// English prompts: the rules it states are Russian grammar (ты/тебя, feminine
|
||||
@@ -61,7 +61,7 @@ func (f Facts) Block(now time.Time) string {
|
||||
}
|
||||
|
||||
b.WriteString(fmt.Sprintf("Сейчас: %s, %d %s %d, %02d:%02d (местное время).\n",
|
||||
ruWeekdays[int(now.Weekday())], now.Day(), ruMonths[int(now.Month())-1], now.Year(),
|
||||
lexicon.Weekday(int(now.Weekday())), now.Day(), lexicon.MonthGenitive(int(now.Month())), now.Year(),
|
||||
now.Hour(), now.Minute()))
|
||||
|
||||
b.WriteString("Умеешь: " + strings.Join(f.can(), "; ") +
|
||||
|
||||
@@ -35,16 +35,14 @@ var dayPlanWords = []string{
|
||||
// answer today and stamp it with today's date, which is a wrong answer where
|
||||
// falling through is only a terse one.
|
||||
//
|
||||
// The weekday names are here as a refusal, not as a feature. "какие планы на
|
||||
// понедельник?" carries no other-day token in the сегодня family and does carry
|
||||
// "планы", so the plan used to claim it and recite today.
|
||||
// A weekday is a refusal too, and it is not in this list: IsDayPlanQuery asks
|
||||
// WeekdayIndex, so every case of every name refuses rather than the nine forms
|
||||
// that used to be written out here (V-581). "какие планы на понедельник?"
|
||||
// carries no other-day token in the сегодня family and does carry "планы", so
|
||||
// the plan used to claim it and recite today.
|
||||
var otherDayWords = []string{
|
||||
"завтра", "послезавтра", "вчера", "позавчера",
|
||||
"tomorrow", "yesterday",
|
||||
"понедельник", "вторник", "среду", "среда", "четверг", "пятницу", "пятница",
|
||||
"субботу", "суббота", "воскресенье",
|
||||
"понедельника", "вторника", "четверга", "пятницы", "субботы", "воскресенья",
|
||||
"monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday",
|
||||
"неделю", "неделя", "недели", "неделе",
|
||||
"выходные", "выходных", "выходным",
|
||||
"месяц", "месяца", "месяце",
|
||||
@@ -69,6 +67,9 @@ func IsDayPlanQuery(text string) bool {
|
||||
}
|
||||
toks := planTokens(text)
|
||||
for _, t := range toks {
|
||||
if _, ok := WeekdayIndex(t); ok {
|
||||
return false
|
||||
}
|
||||
for _, w := range otherDayWords {
|
||||
if t == w {
|
||||
return false
|
||||
|
||||
@@ -45,10 +45,10 @@ try:
|
||||
now = datetime.fromisoformat(sys.argv[2])
|
||||
# Pre-process: replace Russian time qualifiers with AM/PM.
|
||||
# Handles "9 утра", "10 часов утра", "3 часа дня" etc.
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов)?\s+)?утра\b', r'\1 am', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов)?\s+)?вечера\b', r'\1 pm', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов)?\s+)?дня\b', r'\1 pm', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов)?\s+)?ночи\b', r'\1 am', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов|у|ам)?\s+)?утра\b', r'\1 am', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов|у|ам)?\s+)?вечера\b', r'\1 pm', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов|у|ам)?\s+)?дня\b', r'\1 pm', text, flags=re.IGNORECASE)
|
||||
text = re.sub(r'(\d+)\s+(?:час(?:а|ов|у|ам)?\s+)?ночи\b', r'\1 am', text, flags=re.IGNORECASE)
|
||||
# A bare hour after a preposition is dropped on the floor by dateparser:
|
||||
# "завтра в 7" resolves to tomorrow at the CURRENT clock, and "завтра в 7
|
||||
# часов" is read as seven hours from now. Only a qualifier (already an
|
||||
@@ -56,7 +56,9 @@ try:
|
||||
# English "at 7" fails identically, so both prepositions are rewritten.
|
||||
# "на 9" is the same hour said with the other preposition, and it was not
|
||||
# read at all until V-579: "в 9" set the reminder and "на 9" did not.
|
||||
text = re.sub(r'(?<![\w:])(в|во|на|at)\s+([01]?\d|2[0-3])(?:\s+час(?:а|ов)?)?(?![\d:.\w])',
|
||||
# "к двум часам" is a third preposition and the dative that goes with it,
|
||||
# and it was read as no time at all until V-609.
|
||||
text = re.sub(r'(?<![\w:])(в|во|на|к|ко|at|by)\s+([01]?\d|2[0-3])(?:\s+час(?:а|ов|у|ам)?)?(?![\d:.\w])',
|
||||
lambda m: '%s %02d:00' % (m.group(1), int(m.group(2))), text, flags=re.IGNORECASE)
|
||||
settings = {'PREFER_DATES_FROM': 'future', 'RELATIVE_BASE': now}
|
||||
# Two-step: search_dates finds the date substring in text,
|
||||
|
||||
@@ -27,26 +27,10 @@ var habitMarkers = []string{
|
||||
"typically", "normally",
|
||||
}
|
||||
|
||||
// weekdayWords — every form of a weekday name maven needs to recognise,
|
||||
// including the "по …ам" plural the question is usually phrased in.
|
||||
var weekdayWords = map[string]time.Weekday{
|
||||
"понедельник": time.Monday, "понедельникам": time.Monday,
|
||||
"вторник": time.Tuesday, "вторникам": time.Tuesday,
|
||||
"среда": time.Wednesday, "среду": time.Wednesday, "средам": time.Wednesday,
|
||||
"четверг": time.Thursday, "четвергам": time.Thursday,
|
||||
"пятница": time.Friday, "пятницу": time.Friday, "пятницам": time.Friday,
|
||||
"суббота": time.Saturday, "субботу": time.Saturday, "субботам": time.Saturday,
|
||||
"воскресенье": time.Sunday, "воскресеньям": time.Sunday,
|
||||
"воскресенья": time.Sunday, "воскресенью": time.Sunday,
|
||||
"воскресеньем": time.Sunday, "воскресеньях": time.Sunday,
|
||||
"monday": time.Monday, "mondays": time.Monday,
|
||||
"tuesday": time.Tuesday, "tuesdays": time.Tuesday,
|
||||
"wednesday": time.Wednesday, "wednesdays": time.Wednesday,
|
||||
"thursday": time.Thursday, "thursdays": time.Thursday,
|
||||
"friday": time.Friday, "fridays": time.Friday,
|
||||
"saturday": time.Saturday, "saturdays": time.Saturday,
|
||||
"sunday": time.Sunday, "sundays": time.Sunday,
|
||||
}
|
||||
// The weekday a habit question names comes from WeekdayIndex, not from a map
|
||||
// here. This file used to keep its own declension table, which had "воскресеньях"
|
||||
// and no "средах" — a list of forms is finished by whoever last thought of one,
|
||||
// and a dictionary is not (V-581).
|
||||
|
||||
// weekendWords — the weekend as one unit. "что я обычно делаю по выходным?"
|
||||
// has a habit marker and names days, but no weekday name is in it, so it used
|
||||
@@ -77,7 +61,7 @@ func ParseHabitQuery(text string) (HabitQuery, bool) {
|
||||
return HabitQuery{}, false
|
||||
}
|
||||
for _, t := range toks {
|
||||
if wd, ok := weekdayWords[t]; ok {
|
||||
if wd, ok := WeekdayIndex(t); ok {
|
||||
return HabitQuery{Weekday: wd, HasWeekday: true}, true
|
||||
}
|
||||
if weekendWords[t] {
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
package router
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestDativePluralHourIsAnHour — "напомни к двум часам позвонить маме" reached
|
||||
// the daemon with no time at all and she asked the open "Когда?", while "к трём"
|
||||
// one word over read fine (V-609). The word that lost it was "часам", the dative
|
||||
// plural of "час", which four separate hour sets in this package left out.
|
||||
func TestDativePluralHourIsAnHour(t *testing.T) {
|
||||
const s = "напомни к двум часам позвонить маме"
|
||||
if !MentionsTime(s) {
|
||||
t.Errorf("MentionsTime(%q) = false; the sentence names two o'clock", s)
|
||||
}
|
||||
if !NamesAnHour(s) {
|
||||
t.Errorf("NamesAnHour(%q) = false; the sentence names two o'clock", s)
|
||||
}
|
||||
if got, want := SpellOutDigits(s), "напомни к 2 часам позвонить маме"; got != want {
|
||||
t.Errorf("SpellOutDigits(%q) = %q, want %q", s, got, want)
|
||||
}
|
||||
// The slot itself, which is what the daemon reads. It was empty, so
|
||||
// whenGapOf named the hour missing and she asked "Когда?".
|
||||
now := time.Date(2026, 8, 6, 3, 39, 0, 0, time.UTC)
|
||||
ex := Extractor{Time: StubDateTimeParser{}}
|
||||
got := ex.Extract(context.Background(), IntentReminder, s, now)
|
||||
if !got.HasTime {
|
||||
t.Fatalf("the hour was spoken, so the slot must be filled: %+v", got)
|
||||
}
|
||||
if h := got.Time.Hour(); h != 2 && h != 14 {
|
||||
t.Errorf("fire time = %s, want two o'clock in one half of the day or the other", got.Time.Format("15:04"))
|
||||
}
|
||||
}
|
||||
|
||||
// TestHourUnitReachesEverySite — the four sets that read the hour noun now read
|
||||
// one lexicon key, so a form added there is a form all four know. "часам" is the
|
||||
// form that was missing from every one of them.
|
||||
func TestHourUnitReachesEverySite(t *testing.T) {
|
||||
for _, w := range []string{"час", "часа", "часов", "часу", "часам"} {
|
||||
if !numeralContext[w] {
|
||||
t.Errorf("numeralContext is missing %q", w)
|
||||
}
|
||||
if !hourMarkers[w] {
|
||||
t.Errorf("hourMarkers is missing %q", w)
|
||||
}
|
||||
if !timeMarkers[w] {
|
||||
t.Errorf("timeMarkers is missing %q", w)
|
||||
}
|
||||
if _, ok := unitToDuration(2, w); !ok {
|
||||
t.Errorf("unitToDuration does not know %q", w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestMinuteUnitHasTheSameForms — the same defect one noun over: "минутам" was
|
||||
// missing everywhere "минут" and "минуты" were present.
|
||||
func TestMinuteUnitHasTheSameForms(t *testing.T) {
|
||||
for _, w := range []string{"минут", "минуты", "минуту", "минутам"} {
|
||||
if !numeralContext[w] {
|
||||
t.Errorf("numeralContext is missing %q", w)
|
||||
}
|
||||
if !hourMarkers[w] {
|
||||
t.Errorf("hourMarkers is missing %q", w)
|
||||
}
|
||||
if !timeMarkers[w] {
|
||||
t.Errorf("timeMarkers is missing %q", w)
|
||||
}
|
||||
if _, ok := unitToDuration(20, w); !ok {
|
||||
t.Errorf("unitToDuration does not know %q", w)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -34,12 +34,21 @@ func numeralDigit(word string) (string, bool) {
|
||||
// numeralContext — the words that make a numeral a time. A numeral is only
|
||||
// rewritten when one of these sits next to it, so "три яблока" in a note is
|
||||
// left alone and "в три часа" is not.
|
||||
var numeralContext = map[string]bool{
|
||||
"в": true, "во": true, "к": true, "около": true, "на": true,
|
||||
"часа": true, "часов": true, "час": true, "часу": true,
|
||||
"утра": true, "вечера": true, "дня": true, "ночи": true,
|
||||
"минут": true, "минуты": true, "минуту": true,
|
||||
"at": true, "by": true,
|
||||
var numeralContext = buildNumeralContext()
|
||||
|
||||
func buildNumeralContext() map[string]bool {
|
||||
m := map[string]bool{
|
||||
"в": true, "во": true, "к": true, "около": true, "на": true,
|
||||
"утра": true, "вечера": true, "дня": true, "ночи": true,
|
||||
"at": true, "by": true,
|
||||
}
|
||||
for _, w := range lexicon.HourUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
for _, w := range lexicon.MinuteUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// SpellOutDigits rewrites spoken numbers as digits so the date parsers see the
|
||||
|
||||
@@ -175,10 +175,14 @@ func afterWord(s, w string) string {
|
||||
return ""
|
||||
}
|
||||
|
||||
// hourPrepositions — the words a spoken hour sits behind. Three, and no more:
|
||||
// hourPrepositions — the words a spoken hour sits behind. Five, and no more:
|
||||
// the lexicon's frame set is much wider, and a word goes in here only when the
|
||||
// number after it is an hour of the day rather than a count of anything.
|
||||
var hourPrepositions = map[string]bool{"в": true, "во": true, "на": true}
|
||||
//
|
||||
// "к" and "ко" joined the three on V-609. "напомни к двум часам" named an hour
|
||||
// and parsed to nothing, so the reminder reached the daemon with no time and she
|
||||
// asked the open question about an hour he had just said.
|
||||
var hourPrepositions = map[string]bool{"в": true, "во": true, "на": true, "к": true, "ко": true}
|
||||
|
||||
// StubDateTimeParser — a tiny relative/absolute parser standing in for
|
||||
// `dateparser` until the i18n module lands. Handles "in Nh"/"in Nm"/"in Ns" and
|
||||
@@ -398,10 +402,18 @@ func leadingWordNumber(s string) (int, string, bool) {
|
||||
}
|
||||
|
||||
func unitToDuration(n int, unit string) (time.Duration, bool) {
|
||||
switch unit {
|
||||
case "h", "hour", "hours", "hr", "hrs":
|
||||
// The hour and the minute nouns are closed classes with one home in the
|
||||
// lexicon, and the list here used to be short of the oblique forms (V-609).
|
||||
if lexicon.IsHourUnit(unit) {
|
||||
return time.Duration(n) * time.Hour, true
|
||||
case "m", "min", "mins", "minute", "minutes":
|
||||
}
|
||||
if lexicon.IsMinuteUnit(unit) {
|
||||
return time.Duration(n) * time.Minute, true
|
||||
}
|
||||
switch unit {
|
||||
case "h", "hr", "hrs":
|
||||
return time.Duration(n) * time.Hour, true
|
||||
case "m", "min", "mins":
|
||||
return time.Duration(n) * time.Minute, true
|
||||
case "s", "sec", "secs", "second", "seconds":
|
||||
return time.Duration(n) * time.Second, true
|
||||
@@ -409,10 +421,6 @@ func unitToDuration(n int, unit string) (time.Duration, bool) {
|
||||
case "day", "days":
|
||||
return time.Duration(n) * 24 * time.Hour, true
|
||||
// Russian units (inflected forms)
|
||||
case "час", "часа", "часов":
|
||||
return time.Duration(n) * time.Hour, true
|
||||
case "минута", "минуты", "минут":
|
||||
return time.Duration(n) * time.Minute, true
|
||||
case "день", "дня", "дней":
|
||||
return time.Duration(n) * 24 * time.Hour, true
|
||||
case "неделя", "недели", "недель":
|
||||
|
||||
@@ -3,6 +3,7 @@ package router
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/lexicon"
|
||||
"github.com/kami/maven/internal/morph"
|
||||
@@ -205,12 +206,16 @@ var hourMarkers = buildHourMarkers()
|
||||
func buildHourMarkers() map[string]bool {
|
||||
m := map[string]bool{
|
||||
"утра": true, "вечера": true, "дня": true, "ночи": true,
|
||||
"часа": true, "часов": true, "час": true, "часу": true,
|
||||
"минут": true, "минуты": true, "минуту": true,
|
||||
"через": true, "спустя": true, "полчаса": true,
|
||||
"полдень": true, "полночь": true,
|
||||
"am": true, "pm": true, "noon": true, "midnight": true, "in": true,
|
||||
}
|
||||
for _, w := range lexicon.HourUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
for _, w := range lexicon.MinuteUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
for _, w := range lexicon.PartsOfDay() {
|
||||
m[w] = true
|
||||
}
|
||||
@@ -234,16 +239,34 @@ func isMonth(tok string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// isWeekday reports whether the token is a day of the week in any case. The
|
||||
// lexicon lists the nominative, and "в пятницу" is what a reminder says, so the
|
||||
// match is by lemma — grammar is morph's job, not a second word list.
|
||||
func isWeekday(tok string) bool {
|
||||
for i := 0; i < 7; i++ {
|
||||
if morph.SameWord(tok, lexicon.Weekday(i)) {
|
||||
return true
|
||||
// WeekdayIndex reports which day of the week a token names, in any case and in
|
||||
// either language, or false when it names none.
|
||||
//
|
||||
// One matcher for the whole daemon (V-581). Four files used to keep a weekday
|
||||
// list of their own and each one was short in a different direction: the habit
|
||||
// map had "воскресеньях" but no "средах", the plan refusal had "среду" but not
|
||||
// "среде", and cmd/mavend matched the STEM "сред" with strings.Contains, so
|
||||
// "среди" and "средство" read as Wednesday. The lexicon lists the nominative,
|
||||
// every Russian case lemmatises to it, and only English needs its forms written
|
||||
// out — the vendored dictionary is Russian and leaves "mondays" alone.
|
||||
func WeekdayIndex(tok string) (time.Weekday, bool) {
|
||||
t := strings.ToLower(strings.TrimSpace(tok))
|
||||
if n, ok := lexicon.WeekdayEnglish(t); ok {
|
||||
return time.Weekday(n), true
|
||||
}
|
||||
for i, name := range lexicon.Weekdays() {
|
||||
if morph.SameWord(t, name) {
|
||||
return time.Weekday(i), true
|
||||
}
|
||||
}
|
||||
return false
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// isWeekday reports whether the token is a day of the week, when the caller
|
||||
// does not need to know which one.
|
||||
func isWeekday(tok string) bool {
|
||||
_, ok := WeekdayIndex(tok)
|
||||
return ok
|
||||
}
|
||||
|
||||
// timeMarkers — the words that name a time on their own: the qualifiers that
|
||||
@@ -255,11 +278,15 @@ var timeMarkers = buildTimeMarkers()
|
||||
func buildTimeMarkers() map[string]bool {
|
||||
m := map[string]bool{
|
||||
"утра": true, "вечера": true, "дня": true, "ночи": true,
|
||||
"часа": true, "часов": true, "час": true, "часу": true,
|
||||
"минут": true, "минуты": true, "минуту": true,
|
||||
"через": true, "полчаса": true, "сейчас": true,
|
||||
"am": true, "pm": true, "noon": true, "midnight": true,
|
||||
}
|
||||
for _, w := range lexicon.HourUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
for _, w := range lexicon.MinuteUnits() {
|
||||
m[w] = true
|
||||
}
|
||||
for _, w := range lexicon.PartsOfDay() {
|
||||
m[w] = true
|
||||
}
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
package router
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestWeekdayIndexReplacesFourLists — four files kept a weekday list of their
|
||||
// own and each was short in a different direction (V-581). The forms below are
|
||||
// the ones at least one of those lists missed, so they are the point of having
|
||||
// one matcher: the lexicon names the day and the dictionary answers the case.
|
||||
func TestWeekdayIndexReplacesFourLists(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
word string
|
||||
want time.Weekday
|
||||
}{
|
||||
{"понедельник", time.Monday},
|
||||
{"понедельникам", time.Monday},
|
||||
{"понедельником", time.Monday},
|
||||
{"вторник", time.Tuesday},
|
||||
{"среда", time.Wednesday},
|
||||
{"среду", time.Wednesday},
|
||||
{"среде", time.Wednesday},
|
||||
{"средам", time.Wednesday},
|
||||
{"четверга", time.Thursday},
|
||||
{"пятницу", time.Friday},
|
||||
{"субботам", time.Saturday},
|
||||
{"воскресеньях", time.Sunday},
|
||||
{"Воскресенье", time.Sunday},
|
||||
{"monday", time.Monday},
|
||||
{"Fridays", time.Friday},
|
||||
} {
|
||||
got, ok := WeekdayIndex(tc.word)
|
||||
if !ok || got != tc.want {
|
||||
t.Errorf("WeekdayIndex(%q) = %v, %v; want %v, true", tc.word, got, ok, tc.want)
|
||||
}
|
||||
}
|
||||
// A stem match said yes to all of these. A word match says no.
|
||||
for _, w := range []string{"среди", "средство", "средний", "среднем", "субботник", "", "через"} {
|
||||
if _, ok := WeekdayIndex(w); ok {
|
||||
t.Errorf("WeekdayIndex(%q) claimed a weekday", w)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -141,7 +141,8 @@ func LoadSummaries(src rand.Source) (*Summaries, error) {
|
||||
{PlanUncertain, "{line}"},
|
||||
{TasksFirst, "{items}"}, {TasksCandidates, "{items}"},
|
||||
{StallOverdue, "{n}"}, {StallOverdue, "{word}"},
|
||||
{StallSitting, "{n}"}, {StallSitting, "{days}"},
|
||||
{StallSitting, "{n}"}, {StallSitting, "{word}"},
|
||||
{StallSitting, "{days}"}, {StallSitting, "{dayword}"},
|
||||
{StallUnconfirmed, "{n}"}, {StallUnconfirmed, "{word}"},
|
||||
{ReasonOverdueDays, "{n}"}, {ReasonOverdueDays, "{word}"},
|
||||
{ReasonInDays, "{n}"}, {ReasonInDays, "{word}"},
|
||||
|
||||
@@ -75,8 +75,8 @@ func (r *Recognizer) Identify(ctx context.Context, a audio.Audio) (Match, error)
|
||||
if !a.Format.IsValid() {
|
||||
return Match{}, fmt.Errorf("%w: %+v", ErrBadFormat, a.Format)
|
||||
}
|
||||
if seconds(a) < r.minSec {
|
||||
return Match{}, fmt.Errorf("%w: %.1fs, need %.1fs", ErrTooShort, seconds(a), r.minSec)
|
||||
if sec := seconds(a); sec < r.minSec {
|
||||
return Match{}, fmt.Errorf("%w: %.1fs, need %.1fs", ErrTooShort, sec, r.minSec)
|
||||
}
|
||||
vec, err := r.embed(ctx, a)
|
||||
if err != nil {
|
||||
|
||||
@@ -55,17 +55,15 @@ func (s *Store) LastSent(ctx context.Context, key string) (time.Time, error) {
|
||||
// alarm (voice acknowledgment, Telegram callback, etc.).
|
||||
func (s *Store) MarkAcked(ctx context.Context, key string) error {
|
||||
now := time.Now()
|
||||
res, err := s.db.ExecContext(ctx,
|
||||
// No pending nudges is not an error — already acked or never sent — so the
|
||||
// rows-affected count is not read at all: every outcome below this line is
|
||||
// the same nil.
|
||||
_, err := s.db.ExecContext(ctx,
|
||||
`UPDATE nudges SET outcome = 'acted', outcome_ts = ?
|
||||
WHERE rule = ? AND channel = 'telegram' AND outcome = 'pending'`,
|
||||
now.UnixMilli(), key)
|
||||
if err != nil {
|
||||
return fmt.Errorf("mark acked %s: %w", key, err)
|
||||
}
|
||||
n, _ := res.RowsAffected()
|
||||
if n == 0 {
|
||||
// no pending nudges — already acked or never sent; not an error.
|
||||
return nil
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -2,7 +2,6 @@ package store
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
@@ -103,30 +102,16 @@ func (s *Store) ReembedAll(ctx context.Context, currentID string, embed EmbedFun
|
||||
id, text, kind string
|
||||
}
|
||||
var vecs []vecRow
|
||||
rows, err = tx.QueryContext(ctx, `SELECT id, meta FROM memory_vectors`)
|
||||
memRows, err := allMemVectorMetas(ctx, tx)
|
||||
if err != nil {
|
||||
return res, fmt.Errorf("reembed: read memory vectors: %w", err)
|
||||
return res, fmt.Errorf("reembed: %w", err)
|
||||
}
|
||||
for rows.Next() {
|
||||
var id, metaJSON string
|
||||
if err := rows.Scan(&id, &metaJSON); err != nil {
|
||||
rows.Close()
|
||||
return res, fmt.Errorf("reembed: memory row: %w", err)
|
||||
}
|
||||
meta := map[string]string{}
|
||||
if err := json.Unmarshal([]byte(metaJSON), &meta); err != nil {
|
||||
rows.Close()
|
||||
return res, fmt.Errorf("reembed: meta for %q: %w", id, err)
|
||||
}
|
||||
if meta["text"] == "" {
|
||||
for _, v := range memRows {
|
||||
if v.Meta["text"] == "" {
|
||||
res.NoText++
|
||||
continue
|
||||
}
|
||||
vecs = append(vecs, vecRow{id: id, text: meta["text"], kind: meta["type"]})
|
||||
}
|
||||
rows.Close()
|
||||
if err := rows.Err(); err != nil {
|
||||
return res, fmt.Errorf("reembed: memory vectors: %w", err)
|
||||
vecs = append(vecs, vecRow{id: v.ID, text: v.Meta["text"], kind: v.Meta["type"]})
|
||||
}
|
||||
|
||||
for _, v := range vecs {
|
||||
|
||||
@@ -64,9 +64,9 @@ func (s *Store) RepairFactVectors(ctx context.Context, embed EmbedFunc) (FactVec
|
||||
return res, nil
|
||||
}
|
||||
|
||||
rows, err := s.db.QueryContext(ctx, `SELECT id, meta FROM memory_vectors`)
|
||||
memRows, err := allMemVectorMetas(ctx, s.db)
|
||||
if err != nil {
|
||||
return res, fmt.Errorf("repair fact vectors: read: %w", err)
|
||||
return res, fmt.Errorf("repair fact vectors: %w", err)
|
||||
}
|
||||
type factVec struct {
|
||||
id, key string
|
||||
@@ -75,33 +75,19 @@ func (s *Store) RepairFactVectors(ctx context.Context, embed EmbedFunc) (FactVec
|
||||
}
|
||||
var vecs []factVec
|
||||
newest := map[string]int64{} // key → newest ts seen for it
|
||||
for rows.Next() {
|
||||
var id, metaJSON string
|
||||
if err := rows.Scan(&id, &metaJSON); err != nil {
|
||||
rows.Close()
|
||||
return res, fmt.Errorf("repair fact vectors: row: %w", err)
|
||||
}
|
||||
meta := map[string]string{}
|
||||
if err := json.Unmarshal([]byte(metaJSON), &meta); err != nil {
|
||||
rows.Close()
|
||||
return res, fmt.Errorf("repair fact vectors: meta for %q: %w", id, err)
|
||||
}
|
||||
if meta["type"] != "fact" {
|
||||
for _, v := range memRows {
|
||||
if v.Meta["type"] != "fact" {
|
||||
continue
|
||||
}
|
||||
key, ts, ok := splitFactVectorID(id)
|
||||
key, ts, ok := splitFactVectorID(v.ID)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
vecs = append(vecs, factVec{id: id, key: key, meta: meta, ts: ts})
|
||||
vecs = append(vecs, factVec{id: v.ID, key: key, meta: v.Meta, ts: ts})
|
||||
if ts > newest[key] {
|
||||
newest[key] = ts
|
||||
}
|
||||
}
|
||||
rows.Close()
|
||||
if err := rows.Err(); err != nil {
|
||||
return res, fmt.Errorf("repair fact vectors: rows: %w", err)
|
||||
}
|
||||
|
||||
for _, v := range vecs {
|
||||
drop := v.ts < newest[v.key]
|
||||
|
||||
@@ -169,6 +169,49 @@ func (m *MemoryStore) DeletePrefix(ctx context.Context, prefix string) (int64, e
|
||||
return n, nil
|
||||
}
|
||||
|
||||
// memVectorRow is one memory_vectors row with its meta blob decoded — the
|
||||
// shape both ReembedAll (backfill.go) and RepairFactVectors (factvectors.go)
|
||||
// read the whole table as, before each decides what to do with a row on its
|
||||
// own terms (one keys off meta["text"], the other off meta["type"] and the
|
||||
// id's embedded key/timestamp). Query-then-scan was duplicated across the two
|
||||
// before this, id-for-id.
|
||||
type memVectorRow struct {
|
||||
ID string
|
||||
Meta map[string]string
|
||||
}
|
||||
|
||||
// queryContexter is the common surface *sql.DB and *sql.Tx share that
|
||||
// allMemVectorMetas needs. ReembedAll reads inside a transaction so its
|
||||
// migration is atomic; RepairFactVectors reads directly off the db handle.
|
||||
type queryContexter interface {
|
||||
QueryContext(ctx context.Context, query string, args ...any) (*sql.Rows, error)
|
||||
}
|
||||
|
||||
// allMemVectorMetas reads every memory_vectors row and decodes its meta blob.
|
||||
func allMemVectorMetas(ctx context.Context, q queryContexter) ([]memVectorRow, error) {
|
||||
rows, err := q.QueryContext(ctx, `SELECT id, meta FROM memory_vectors`)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read memory vectors: %w", err)
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []memVectorRow
|
||||
for rows.Next() {
|
||||
var id, metaJSON string
|
||||
if err := rows.Scan(&id, &metaJSON); err != nil {
|
||||
return nil, fmt.Errorf("memory vector row: %w", err)
|
||||
}
|
||||
meta := map[string]string{}
|
||||
if err := json.Unmarshal([]byte(metaJSON), &meta); err != nil {
|
||||
return nil, fmt.Errorf("meta for %q: %w", id, err)
|
||||
}
|
||||
out = append(out, memVectorRow{ID: id, Meta: meta})
|
||||
}
|
||||
if err := rows.Err(); err != nil {
|
||||
return nil, fmt.Errorf("memory vectors: %w", err)
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// escapeLike neutralises the LIKE wildcards in a literal prefix.
|
||||
func escapeLike(s string) string {
|
||||
r := strings.NewReplacer(`\`, `\\`, `%`, `\%`, `_`, `\_`)
|
||||
|
||||
@@ -187,7 +187,10 @@ func (s *Store) AcceptProposedRoutine(ctx context.Context, id int64, ts time.Tim
|
||||
if err != nil {
|
||||
return fmt.Errorf("accept proposed routine: %w", err)
|
||||
}
|
||||
n, _ := res.RowsAffected()
|
||||
n, err := res.RowsAffected()
|
||||
if err != nil {
|
||||
return fmt.Errorf("accept proposed routine: rows affected: %w", err)
|
||||
}
|
||||
if n == 0 {
|
||||
return fmt.Errorf("%w: id=%d not in 'proposed' status", ErrProposedRoutineNotFound, id)
|
||||
}
|
||||
|
||||
@@ -23,6 +23,9 @@ func TestLexiconRewritesNames(t *testing.T) {
|
||||
// Two names in a row share the space between them, which one pass
|
||||
// would consume.
|
||||
{"GPU GPU", "джи-пи-ю джи-пи-ю"},
|
||||
// Two passes cover a run of any length, because the first pass takes
|
||||
// every other name and leaves both boundaries of the ones it skipped.
|
||||
{"GPU GPU GPU GPU", "джи-пи-ю джи-пи-ю джи-пи-ю джи-пи-ю"},
|
||||
// Not a word boundary: a name inside a longer token is left alone.
|
||||
{"vikunjaless", "vikunjaless"},
|
||||
{"ничего не совпало", "ничего не совпало"},
|
||||
|
||||
+9
-5
@@ -20,6 +20,7 @@ package tts
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"math"
|
||||
|
||||
@@ -50,17 +51,20 @@ func NewStub() *Stub { return &Stub{} }
|
||||
// silent no-op a bug could hide behind).
|
||||
func (s *Stub) Synthesize(_ context.Context, text string) (audio.Audio, error) {
|
||||
const durMs = 200
|
||||
const samples = 16000 * durMs / 1000 // 3200 samples @ 16k
|
||||
pcm := make([]byte, samples*2)
|
||||
// The rate is read off the canonical format rather than written again, so
|
||||
// the tone stays in tune with the shape the seam declares.
|
||||
rate := audio.PCM16kMono.SampleRate
|
||||
bytesPerSample := audio.PCM16kMono.SampleBits / 8
|
||||
samples := rate * durMs / 1000
|
||||
pcm := make([]byte, samples*bytesPerSample)
|
||||
freq := 220.0 // A3
|
||||
if len(text) > 0 {
|
||||
freq = 180.0 + float64(text[0]%6)*60 // 180..480 Hz band
|
||||
}
|
||||
for i := 0; i < samples; i++ {
|
||||
t := float64(i) / 16000.0
|
||||
t := float64(i) / float64(rate)
|
||||
v := int16(12000 * math.Sin(2*math.Pi*freq*t))
|
||||
pcm[i*2] = byte(v)
|
||||
pcm[i*2+1] = byte(v >> 8)
|
||||
binary.LittleEndian.PutUint16(pcm[i*2:], uint16(v))
|
||||
}
|
||||
return audio.Audio{Format: audio.PCM16kMono, Bytes: pcm}, nil
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"strings"
|
||||
|
||||
"github.com/kami/maven/internal/lexicon"
|
||||
"github.com/kami/maven/internal/say"
|
||||
)
|
||||
|
||||
// The month names are a closed class and live in internal/lexicon, 1-indexed,
|
||||
@@ -32,7 +33,7 @@ func Speakable(s string) string {
|
||||
})
|
||||
s = reTime.ReplaceAllStringFunc(s, func(m string) string {
|
||||
p := reTime.FindStringSubmatch(m)
|
||||
return p[1] + " часов " + p[2] + " минут"
|
||||
return spokenTime(mustInt(p[1]), mustInt(p[2]))
|
||||
})
|
||||
s = reDots.ReplaceAllStringFunc(s, func(m string) string {
|
||||
return strings.Join(strings.Split(m, "."), " точка ")
|
||||
@@ -46,6 +47,20 @@ func Speakable(s string) string {
|
||||
return Pronounce(s)
|
||||
}
|
||||
|
||||
// spokenTime reads a clock time the way it is said rather than the way it is
|
||||
// written. Two things the written form gets wrong out loud. The noun after a
|
||||
// numeral inflects, so 21:00 is "час" and 22:00 is "часа", where the old
|
||||
// rewrite said "часов" for every hour and "минут" for every minute. And a
|
||||
// leading zero is punctuation, not a word: 14:00 is "14 часов" and 9:05 is
|
||||
// "9 часов 5 минут", never "00 минут" or "05 минут".
|
||||
func spokenTime(h, m int) string {
|
||||
out := strconv.Itoa(h) + " " + say.CountWord(h, "час", "часа", "часов")
|
||||
if m == 0 {
|
||||
return out
|
||||
}
|
||||
return out + " " + strconv.Itoa(m) + " " + say.CountWord(m, "минута", "минуты", "минут")
|
||||
}
|
||||
|
||||
func spokenDate(dd, mm, yyyy string) string {
|
||||
mi, _ := strconv.Atoi(mm)
|
||||
if mi < 1 || mi > 12 {
|
||||
|
||||
@@ -6,8 +6,11 @@ func TestSpeakable(t *testing.T) {
|
||||
cases := []struct{ in, want string }{
|
||||
{"напомню 10.07.2026", "напомню 10 июля 2026"},
|
||||
{"срок 01.01", "срок 1 января"},
|
||||
{"встреча в 14:00", "встреча в 14 часов 00 минут"},
|
||||
{"в 9:05 подъём", "в 9 часов 05 минут подъём"},
|
||||
{"встреча в 14:00", "встреча в 14 часов"},
|
||||
{"в 9:05 подъём", "в 9 часов 5 минут подъём"},
|
||||
{"в 21:00 отбой", "в 21 час отбой"},
|
||||
{"в 22:02 отбой", "в 22 часа 2 минуты отбой"},
|
||||
{"в 1:01 проснулся", "в 1 час 1 минута проснулся"},
|
||||
{"это 3.2.1 версия", "это 3 точка 2 точка 1 версия"},
|
||||
{"без чисел", "без чисел"},
|
||||
}
|
||||
|
||||
@@ -14,11 +14,13 @@ import (
|
||||
"crypto/elliptic"
|
||||
"crypto/rand"
|
||||
"crypto/sha256"
|
||||
"crypto/subtle"
|
||||
"encoding/base64"
|
||||
"encoding/binary"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"math/big"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
@@ -56,8 +58,16 @@ type credentialAssertion struct {
|
||||
|
||||
// RP — the relying party instance. Holds config and transient challenge state.
|
||||
// A single-user daemon has one RP.
|
||||
//
|
||||
// mavweb serves the four passkey endpoints from its HTTP handlers, so the two
|
||||
// challenge maps are reached concurrently even on a single-user box: a browser
|
||||
// retrying an assertion while another tab begins one is enough. A concurrent
|
||||
// map write is a fatal runtime error, not a recovered panic, so it would take
|
||||
// the whole daemon down from an endpoint that answers before any credential is
|
||||
// proven. Every read and write of regs and asserts is under mu.
|
||||
type RP struct {
|
||||
cfg Config
|
||||
mu sync.Mutex
|
||||
regs map[string]*credentialRegistration
|
||||
asserts map[string]*credentialAssertion
|
||||
challengeTTL time.Duration
|
||||
@@ -75,6 +85,13 @@ func NewRP(cfg Config) *RP {
|
||||
|
||||
// CleanExpired removes challenges older than the TTL.
|
||||
func (rp *RP) CleanExpired() {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
rp.cleanExpired()
|
||||
}
|
||||
|
||||
// cleanExpired is CleanExpired for a caller that already holds mu.
|
||||
func (rp *RP) cleanExpired() {
|
||||
now := time.Now()
|
||||
for k, r := range rp.regs {
|
||||
if now.Sub(r.CreatedAt) > rp.challengeTTL {
|
||||
@@ -97,12 +114,14 @@ func (rp *RP) CreationOptions(userID []byte, userName string) (map[string]any, s
|
||||
}
|
||||
challengeB64 := base64.RawURLEncoding.EncodeToString(challenge)
|
||||
|
||||
rp.CleanExpired()
|
||||
rp.mu.Lock()
|
||||
rp.cleanExpired()
|
||||
rp.regs[challengeB64] = &credentialRegistration{
|
||||
Challenge: challengeB64,
|
||||
UserID: userID,
|
||||
CreatedAt: time.Now(),
|
||||
}
|
||||
rp.mu.Unlock()
|
||||
|
||||
return map[string]any{
|
||||
"rp": map[string]string{
|
||||
@@ -136,12 +155,10 @@ func (rp *RP) CreationOptions(userID []byte, userName string) (map[string]any, s
|
||||
|
||||
// FinishRegistration parses the browser's response and stores the credential.
|
||||
func (rp *RP) FinishRegistration(save CredentialSaver, challengeB64 string, resp map[string]any) (string, error) {
|
||||
rp.CleanExpired()
|
||||
reg, ok := rp.regs[challengeB64]
|
||||
reg, ok := rp.takeReg(challengeB64)
|
||||
if !ok {
|
||||
return "", fmt.Errorf("webauthn: unknown or expired challenge")
|
||||
}
|
||||
delete(rp.regs, challengeB64)
|
||||
|
||||
credID := rawString(resp, "id")
|
||||
if credID == "" {
|
||||
@@ -188,11 +205,13 @@ func (rp *RP) AssertionOptions() (map[string]any, string, error) {
|
||||
}
|
||||
challengeB64 := base64.RawURLEncoding.EncodeToString(challenge)
|
||||
|
||||
rp.CleanExpired()
|
||||
rp.mu.Lock()
|
||||
rp.cleanExpired()
|
||||
rp.asserts[challengeB64] = &credentialAssertion{
|
||||
Challenge: challengeB64,
|
||||
CreatedAt: time.Now(),
|
||||
}
|
||||
rp.mu.Unlock()
|
||||
|
||||
return map[string]any{
|
||||
"challenge": challengeB64,
|
||||
@@ -215,11 +234,9 @@ func (rp *RP) AssertionOptions() (map[string]any, string, error) {
|
||||
// FinishAssertion verifies the browser's assertion response and returns the
|
||||
// verified credential ID.
|
||||
func (rp *RP) FinishAssertion(lookup CredentialLookup, updateSignCount SignCountUpdater, challengeB64 string, resp map[string]any) (string, error) {
|
||||
rp.CleanExpired()
|
||||
if _, ok := rp.asserts[challengeB64]; !ok {
|
||||
if !rp.takeAssert(challengeB64) {
|
||||
return "", fmt.Errorf("webauthn: unknown or expired challenge")
|
||||
}
|
||||
delete(rp.asserts, challengeB64)
|
||||
|
||||
credID := rawString(resp, "id")
|
||||
if credID == "" {
|
||||
@@ -298,6 +315,34 @@ func (rp *RP) FinishAssertion(lookup CredentialLookup, updateSignCount SignCount
|
||||
return credID, nil
|
||||
}
|
||||
|
||||
// takeReg removes and returns the in-flight registration for challengeB64.
|
||||
// Taking under one lock is what makes a challenge single-use: looking it up
|
||||
// and deleting it separately lets two replays of the same response both find
|
||||
// it before either deletes.
|
||||
func (rp *RP) takeReg(challengeB64 string) (*credentialRegistration, bool) {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
rp.cleanExpired()
|
||||
reg, ok := rp.regs[challengeB64]
|
||||
if ok {
|
||||
delete(rp.regs, challengeB64)
|
||||
}
|
||||
return reg, ok
|
||||
}
|
||||
|
||||
// takeAssert removes the in-flight assertion for challengeB64 and reports
|
||||
// whether it was there. Single-use for the same reason takeReg is.
|
||||
func (rp *RP) takeAssert(challengeB64 string) bool {
|
||||
rp.mu.Lock()
|
||||
defer rp.mu.Unlock()
|
||||
rp.cleanExpired()
|
||||
if _, ok := rp.asserts[challengeB64]; !ok {
|
||||
return false
|
||||
}
|
||||
delete(rp.asserts, challengeB64)
|
||||
return true
|
||||
}
|
||||
|
||||
func verifyClientDataBytes(clientDataJSON []byte, expectedType, expectedChallenge, expectedOrigin string) error {
|
||||
var cdj struct {
|
||||
Type string `json:"type"`
|
||||
@@ -310,7 +355,10 @@ func verifyClientDataBytes(clientDataJSON []byte, expectedType, expectedChalleng
|
||||
if cdj.Type != expectedType {
|
||||
return fmt.Errorf("webauthn: unexpected type %q", cdj.Type)
|
||||
}
|
||||
if cdj.Challenge != expectedChallenge {
|
||||
// Constant time, because the challenge is the one secret in clientDataJSON:
|
||||
// it is 32 bytes of crypto/rand the browser has to echo back, and a
|
||||
// byte-at-a-time compare is the shape that leaks a guessed prefix.
|
||||
if subtle.ConstantTimeCompare([]byte(cdj.Challenge), []byte(expectedChallenge)) != 1 {
|
||||
return fmt.Errorf("webauthn: challenge mismatch")
|
||||
}
|
||||
if cdj.Origin != expectedOrigin {
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"encoding/base64"
|
||||
"encoding/binary"
|
||||
"encoding/json"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -179,6 +180,53 @@ func TestRegisterAssertRoundTrip(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The passkey endpoints are HTTP handlers, so two browsers beginning a
|
||||
// challenge at once reach the same RP. Under -race this fails on the bare maps
|
||||
// it used to keep, and in production a concurrent map write is fatal.
|
||||
func TestChallengeMapsAreConcurrencySafe(t *testing.T) {
|
||||
rp := NewRP(Config{Origin: testOrigin, RPID: testRPID, RPName: "maven"})
|
||||
lookup := func(string) ([]byte, int64, error) { return nil, 0, nil }
|
||||
upd := func(string, int64) error { return nil }
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < 16; i++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for j := 0; j < 32; j++ {
|
||||
_, chal, err := rp.AssertionOptions()
|
||||
if err != nil {
|
||||
t.Error(err)
|
||||
return
|
||||
}
|
||||
_, _ = rp.FinishAssertion(lookup, upd, chal, map[string]any{})
|
||||
if _, _, err := rp.CreationOptions([]byte("u"), "user"); err != nil {
|
||||
t.Error(err)
|
||||
return
|
||||
}
|
||||
rp.CleanExpired()
|
||||
}
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// A challenge is single-use: the second presentation of one already spent is
|
||||
// unknown, whichever goroutine gets there first.
|
||||
func TestAssertionChallengeIsSingleUse(t *testing.T) {
|
||||
rp := NewRP(Config{Origin: testOrigin, RPID: testRPID, RPName: "maven"})
|
||||
_, chal, err := rp.AssertionOptions()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !rp.takeAssert(chal) {
|
||||
t.Fatal("first take of a fresh challenge failed")
|
||||
}
|
||||
if rp.takeAssert(chal) {
|
||||
t.Fatal("a spent challenge was accepted twice")
|
||||
}
|
||||
}
|
||||
|
||||
// TestAssertRejectsWrongOrigin — a phished assertion from another origin fails.
|
||||
func TestAssertRejectsWrongOrigin(t *testing.T) {
|
||||
err := verifyClientDataBytes(clientData("webauthn.get", "abc", "https://evil.test"), "webauthn.get", "abc", testOrigin)
|
||||
|
||||
+18
-14
@@ -37,8 +37,9 @@ type Server struct {
|
||||
addr netaddr.Addr
|
||||
ln net.Listener
|
||||
|
||||
wg sync.WaitGroup
|
||||
done chan struct{}
|
||||
wg sync.WaitGroup
|
||||
done chan struct{}
|
||||
closeOnce sync.Once
|
||||
|
||||
// connCount — assigned per accepted conn, used in logs to distinguish
|
||||
// concurrent connections. Monotonic; not load-bearing for correctness.
|
||||
@@ -180,20 +181,23 @@ func (srv *Server) dispatch(ctx context.Context, req Request) (json.RawMessage,
|
||||
}
|
||||
|
||||
// Close stops accepting and waits for in-flight connections to drain. The
|
||||
// socket file is removed so a restart can rebind cleanly. Idempotent.
|
||||
// socket file is removed so a restart can rebind cleanly.
|
||||
//
|
||||
// Idempotent, and safe from two goroutines at once. The check-then-close it
|
||||
// replaced let both callers see an open channel and the second close panicked,
|
||||
// so a shutdown racing a signal handler took the process down the one way a
|
||||
// clean shutdown is supposed to prevent.
|
||||
func (srv *Server) Close() error {
|
||||
select {
|
||||
case <-srv.done:
|
||||
return nil
|
||||
default:
|
||||
var err error
|
||||
srv.closeOnce.Do(func() {
|
||||
close(srv.done)
|
||||
}
|
||||
if srv.ln == nil {
|
||||
return nil
|
||||
}
|
||||
err := srv.ln.Close()
|
||||
srv.wg.Wait()
|
||||
netaddr.Cleanup(srv.addr)
|
||||
if srv.ln == nil {
|
||||
return
|
||||
}
|
||||
err = srv.ln.Close()
|
||||
srv.wg.Wait()
|
||||
netaddr.Cleanup(srv.addr)
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user