Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 4666057066 | |||
| 50c6637c1b | |||
| ee9d55ca95 | |||
| a886217223 | |||
| de9884e063 | |||
| bbefda66e2 | |||
| a37c4138a1 | |||
| d6f391430f |
+12
-2
@@ -489,15 +489,19 @@ func (h *reactiveHandler) noteSuspended(ctx context.Context, q *dialogue.Pending
|
||||
if !q.CanResume() {
|
||||
h.clarifyStore.Delete(dialogueIDOf(ctx))
|
||||
h.noteDropped(ctx)
|
||||
log.Printf("voice: clarify — the question about %s stepped aside %d times; letting the request go", q.Missing[0], q.Suspends)
|
||||
log.Printf("voice: clarify — letting the question about %s go: %d asides in a row, %d rides in all", q.Missing[0], q.Suspends, q.Rides)
|
||||
return
|
||||
}
|
||||
q.Suspends++
|
||||
// Rides is the same event counted without the reset (V-663). Incremented
|
||||
// beside Suspends and never anywhere else, so the two cannot disagree about
|
||||
// what happened, only about how much of it they remember.
|
||||
q.Rides++
|
||||
q.Asked = h.now()
|
||||
h.clarifyStore.Put(dialogueIDOf(ctx), q)
|
||||
rt.resume = question
|
||||
rt.suspended = true
|
||||
log.Printf("voice: clarify — is its own request; suspending the question about %s and resuming it in the same reply (suspend %d of %d)", q.Missing[0], q.Suspends, dialogue.MaxSuspends)
|
||||
log.Printf("voice: clarify — is its own request; suspending the question about %s and resuming it in the same reply (suspend %d of %d, ride %d of %d)", q.Missing[0], q.Suspends, dialogue.MaxSuspends, q.Rides, dialogue.MaxRides)
|
||||
}
|
||||
|
||||
// foldAnswerIntoUtterance appends an answered subject to the original words,
|
||||
@@ -540,6 +544,11 @@ func (h *reactiveHandler) askRemainingGap(ctx context.Context, q *dialogue.Pendi
|
||||
// Suspends is not carried, and by this point it is already zero: the answer
|
||||
// path resets it (V-654). Left off the literal so the zero is stated where
|
||||
// the struct is built, rather than inherited from a field nobody names.
|
||||
//
|
||||
// Rides IS carried, and that is the whole point of it (V-663). This is the
|
||||
// same request under a second question, not a new one, so the turns it has
|
||||
// already ridden still count against it. Dropping the field here is exactly
|
||||
// the re-basing that let one question ride twenty-six replies.
|
||||
h.clarifyStore.Put(dialogueIDOf(ctx), &dialogue.PendingQuestion{
|
||||
Intent: q.Intent,
|
||||
Slots: merged,
|
||||
@@ -550,6 +559,7 @@ func (h *reactiveHandler) askRemainingGap(ctx context.Context, q *dialogue.Pendi
|
||||
TTL: clarifyTTL,
|
||||
Attempts: q.Attempts + 1,
|
||||
MaxAttempts: q.MaxAttempts,
|
||||
Rides: q.Rides,
|
||||
})
|
||||
log.Printf("voice: clarify — one gap filled, still missing %s for intent=%s, asking again (attempt %d)", remaining[0], intent, q.Attempts+1)
|
||||
return question, true
|
||||
|
||||
@@ -186,6 +186,28 @@ func carriesReminderVerb(text string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// isPleasantry matches the WHOLE utterance against lexicon.Pleasantries, after
|
||||
// lowercasing and dropping the punctuation a greeting carries.
|
||||
//
|
||||
// Whole utterance and not tokens. Every token rule tried here was wrong on
|
||||
// something: "вечер" answers "это утра или вечера?", "нет" answers a confirm,
|
||||
// and "спокойной" alone is not an utterance at all. A greeting is a fixed
|
||||
// phrase, so matching it as one costs nothing and claims nothing else.
|
||||
func isPleasantry(text string) bool {
|
||||
t := strings.ToLower(strings.TrimSpace(text))
|
||||
t = strings.Trim(t, " .,!?…")
|
||||
t = strings.Join(strings.Fields(t), " ")
|
||||
if t == "" {
|
||||
return false
|
||||
}
|
||||
for _, p := range lexicon.Pleasantries() {
|
||||
if t == p {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// offlineOwnRequest is the shape half of the evidence: the offline token tests,
|
||||
// which cost nothing and never depend on the model that produced the routing.
|
||||
// It is also the whole answer when there is no route to read — the classifier
|
||||
@@ -225,6 +247,16 @@ func classifyTurnRole(q *dialogue.PendingQuestion, text string, answer dialogue.
|
||||
// hour, and no route saying "question" changes that. It works because the
|
||||
// extractor no longer reads a day word as the current clock, so a sentence
|
||||
// that names no hour now fills nothing to weigh.
|
||||
// A pleasantry is neither (V-663). "спасибо" and "привет" fell through to
|
||||
// roleAnswer, so a question about a reminder's DAY was re-asked at a man
|
||||
// saying thank you, and the retry it spent was one of the three bounds
|
||||
// meant to end the ride. It is an aside: answered as itself, the question
|
||||
// resumed on the tail, no attempt spent, one ride counted. Placed above the
|
||||
// content gate because "доброе утро" has content and states nothing, so
|
||||
// neither half of the evidence below can reach it.
|
||||
if q != nil && isPleasantry(text) {
|
||||
return roleAside
|
||||
}
|
||||
own := false
|
||||
if len(ownContent(text)) > 0 {
|
||||
own = offlineOwnRequest(text) || (ok && carriesOwnRequest(routed, text))
|
||||
|
||||
@@ -372,4 +372,69 @@ func TestAnAnsweredGapResetsTheSuspendBudget(t *testing.T) {
|
||||
if q.Suspends != 0 {
|
||||
t.Fatalf("answering a gap must reset the suspend budget: suspends = %d", q.Suspends)
|
||||
}
|
||||
// The ride it already took is carried across the re-park (V-663). Resetting
|
||||
// both counters here is what let one question ride twenty-six replies.
|
||||
if q.Rides != 1 {
|
||||
t.Fatalf("the aside it already took was forgotten: rides = %d", q.Rides)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTwoBoundsCannotRearmEachOther — V-663.
|
||||
//
|
||||
// MaxSuspends landed and the measurement did not move: twenty-six of 140 turns
|
||||
// carried a tail before it and twenty-six after. This is the shape it misses,
|
||||
// taken from the 2026-08-08 run, where one question rode turns 7 to 13.
|
||||
//
|
||||
// An aside spends no attempt, so MaxAttempts never reaches it. A turn that
|
||||
// reads as a failed answer zeroes Suspends, so MaxSuspends never reaches the
|
||||
// asides either. Alternating the two rearms each bound with the other's
|
||||
// traffic. Rides counts both kinds and is never reset, so it is what ends this.
|
||||
func TestTwoBoundsCannotRearmEachOther(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
h, _ := newRoutingClarifyHandler(t)
|
||||
id := dialogueIDFor(sourceText, "web")
|
||||
resumed, _ := clarifyResumedFor(dialogue.SlotTime)
|
||||
|
||||
if reply := h.handleText(ctx, "web", "напомни позвонить маме"); !strings.Contains(reply, "?") {
|
||||
t.Fatalf("expected the time question, got %q", reply)
|
||||
}
|
||||
|
||||
// Two asides. Each one rides and neither spends an attempt.
|
||||
for i := 0; i < 2; i++ {
|
||||
reply := h.handleText(ctx, "web", "какие у меня напоминания?")
|
||||
if !strings.HasSuffix(reply, resumed) {
|
||||
t.Fatalf("aside %d: the question must come back, got %q", i+1, reply)
|
||||
}
|
||||
}
|
||||
q := h.clarifyStore.Get(id, h.now())
|
||||
if q == nil || q.Rides != 2 || q.Suspends != 2 {
|
||||
t.Fatalf("after two asides: %+v", q)
|
||||
}
|
||||
|
||||
// A pleasantry. It used to read as a failed answer, so she re-asked the
|
||||
// question at a man saying thank you and spent an attempt doing it. Now it
|
||||
// is an aside: answered as itself, question on the tail, one more ride.
|
||||
reply := h.handleText(ctx, "web", "спасибо")
|
||||
if !strings.HasSuffix(reply, resumed) {
|
||||
t.Fatalf("a pleasantry lost the parked question: %q", reply)
|
||||
}
|
||||
q = h.clarifyStore.Get(id, h.now())
|
||||
if q == nil || q.Attempts != 1 {
|
||||
t.Fatalf("a pleasantry spent an attempt: %+v", q)
|
||||
}
|
||||
if q.Rides != 3 {
|
||||
t.Fatalf("a pleasantry rode free: %+v", q)
|
||||
}
|
||||
|
||||
// One more ride of any kind and the request goes, out loud.
|
||||
reply = h.handleText(ctx, "web", "какие у меня напоминания?")
|
||||
if !strings.Contains(reply, clarifyDropped) {
|
||||
t.Fatalf("the question rode four asides and was let go in silence: %q", reply)
|
||||
}
|
||||
if strings.HasSuffix(reply, resumed) {
|
||||
t.Fatalf("a question she has let go must not be asked again: %q", reply)
|
||||
}
|
||||
if h.clarifyStore.Get(id, h.now()) != nil {
|
||||
t.Fatal("the question must be gone once she has said she let it go")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -308,6 +308,29 @@ The count is of CONSECUTIVE step-asides. It resets the moment he answers, in
|
||||
too. "Позвонить маме" against a question about the time is still him in the
|
||||
exchange. The retry it costs is bound enough on its own.
|
||||
|
||||
#### And it may ride four turns in all
|
||||
|
||||
Decided 2026-08-08 (V-663), because the bound above did not move the number it
|
||||
was written for. Twenty-six of 140 turns carried a tail before it landed and
|
||||
twenty-six carried one after.
|
||||
|
||||
Two bounds rearm each other. An aside spends no attempt, so `MaxAttempts` never
|
||||
reaches it. A turn that reads as a failed answer zeroes `Suspends`, so
|
||||
`MaxSuspends` never reaches the asides. Alternating them, each bound is restored
|
||||
by the other's traffic. Measured on 2026-08-08: one question about a reminder's
|
||||
day rode turns 7 to 13. It ended only because turn 14 was a new request.
|
||||
|
||||
`PendingQuestion.Rides` counts the same event as `Suspends` with the resets
|
||||
taken out. It is set once, incremented only in `noteSuspended`, carried across
|
||||
the re-park in `askRemainingGap`, and read by nothing that could lower it.
|
||||
`MaxRides` is 4, one looser than `MaxSuspends` so that the tighter statement
|
||||
about a run stays reachable.
|
||||
|
||||
This is a bound, not a cure. It ends the measured ride one turn early. Most of
|
||||
that ride's length is attempts, spent because `classifyTurnRole` reads "спасибо"
|
||||
and "привет" as failed answers to a question about a day. That is the next
|
||||
thing to fix and it is not a bound.
|
||||
|
||||
The re-ask is also two sentences rather than one. It used to be spliced onto the
|
||||
answer with a comma. On a real answer that buries the question in the tail of
|
||||
one run-on thought:
|
||||
|
||||
@@ -0,0 +1,88 @@
|
||||
# The parked clarify ride, bounded and re-measured
|
||||
|
||||
Date: 2026-08-08, V-663. Same 140 turns, same driver, third and fourth runs of
|
||||
the day. Before is `d6f3914`, after is that plus two changes.
|
||||
|
||||
## What was measured before
|
||||
|
||||
One question about a reminder's day rode turns 6 to 13. It ended only because
|
||||
turn 14 was a new request. Two of those turns are the worst replies in the
|
||||
corpus:
|
||||
|
||||
```text
|
||||
спасибо -> Сейчас 21:25. В какой день?
|
||||
привет -> Сейчас 21:25. В какой день?
|
||||
```
|
||||
|
||||
V-654 had already added `MaxSuspends` and the tail count had not moved.
|
||||
|
||||
## Why three bounds let it happen
|
||||
|
||||
The TTL, `MaxAttempts` and `MaxSuspends` all exist and all were rearmed.
|
||||
|
||||
An aside spends no attempt, so `MaxAttempts` never reaches it. A turn reading as
|
||||
a failed answer zeroes `Suspends`, so `MaxSuspends` never reaches the asides.
|
||||
`спасибо` read as a failed answer. So the pleasantry both produced the reply
|
||||
above and paid for it out of the budget meant to end the ride.
|
||||
|
||||
## The two changes
|
||||
|
||||
`PendingQuestion.Rides` counts the same event as `Suspends` with the resets
|
||||
taken out. Set once, incremented only in `noteSuspended`, carried across the
|
||||
re-park in `askRemainingGap`. `MaxRides` is 4, one looser than `MaxSuspends` so
|
||||
that the tighter statement about a run stays reachable.
|
||||
|
||||
A pleasantry is an aside, not a failed answer. The set is a new closed lexicon
|
||||
entry, matched as whole utterances rather than tokens. `вечер` answers `это утра
|
||||
или вечера?` and `нет` answers a confirm. Anything that could fill a slot stays
|
||||
out.
|
||||
|
||||
## Result
|
||||
|
||||
| | before | after |
|
||||
|---|---|---|
|
||||
| turns carrying a clarify tail | 21 | 17 |
|
||||
| turns carrying any failure string | 31 | 29 |
|
||||
| the longest ride | 8 turns | 4 turns |
|
||||
|
||||
The turns carrying a tail, by number:
|
||||
|
||||
```text
|
||||
before 6 7 8 9 10 11 12 13 53 54 55 56 57 99 102 103 116 136 137 138 139
|
||||
after 6 7 8 9 53 54 55 56 57 99 100 101 116 136 137 138 139
|
||||
```
|
||||
|
||||
Turn 10 is the change. It now reads:
|
||||
|
||||
```text
|
||||
спасибо -> Прошлую просьбу отпускаю. Пожалуйста, я всегда готова помочь тебе.
|
||||
```
|
||||
|
||||
She lets the request go, says so, and answers the man. Turns 11 to 13 are clean.
|
||||
|
||||
**`MaxRides` is not what fired.** The pleasantry is an aside now, so it no
|
||||
longer breaks the run. `MaxSuspends` reached three on turn 10 and ended it.
|
||||
`Rides` is the backstop for the shape where an answer really does break the run.
|
||||
No turn in this corpus reaches it.
|
||||
|
||||
## What did not move
|
||||
|
||||
Four rides are untouched. Turns 53 to 57 are five consecutive asides against a
|
||||
reminder missing its day. Turn 58 is a new request that drops it. Nothing
|
||||
pleasant appears in that run, so neither change applies. Turns 99 to 101 shifted
|
||||
by one, and 116 and 136 to 139 are unchanged.
|
||||
|
||||
So the fix is worth four turns of twenty-one. What is left is asides against a
|
||||
question the owner never answers. `MaxSuspends` was written for that shape and
|
||||
does bound it, at four turns each.
|
||||
|
||||
## Not attributable
|
||||
|
||||
Latency moved p50 1.1s to 1.5s and p95 2.8s to 3.0s, and the 34.3s outlier in
|
||||
the earlier run is gone. Both runs had the workstation up. Read none of it as
|
||||
caused by this change.
|
||||
|
||||
One unrelated defect appeared in the after run and is recorded here because it
|
||||
is visible in the transcript. Turn 4 answered `Я записала твою привычкуRegarding
|
||||
coffee without sugar.` That is English leaking into a Russian reply with no
|
||||
space in front of it. It is a phrasing defect and it has no task yet.
|
||||
@@ -80,12 +80,53 @@ V-655 was never going to touch this. A parked clarify is dialogue state and not
|
||||
a query source. It remains the single worst thing about talking to her. The week test, the
|
||||
fortnight test and this re-run all report it unchanged.
|
||||
|
||||
## A gap in the harness
|
||||
## A gap in the harness, fixed and re-run the same day
|
||||
|
||||
`ipc.ChatReply.Source` came back empty on all 140 turns, in both runs. The
|
||||
driver reads it from the redirect query string and there is nothing there. So
|
||||
the badge that says which query source claimed a turn is invisible to the
|
||||
harness, and every finding above is read off the reply text instead.
|
||||
driver read the redirect parameter `src` and `cmd/mavweb/chat.go` writes `s`.
|
||||
So every finding above is read off the reply text instead of off the badge.
|
||||
|
||||
That is worth fixing before the next re-run. Reading the source directly would
|
||||
have shown the two homelab misses without inferring them from the wording.
|
||||
Fixed in V-662 and the 140 turns were driven a third time. Sixty-eight of them
|
||||
name a source. The rest are not query turns and never reach `queryWalk`.
|
||||
|
||||
| source | turns |
|
||||
|---|---|
|
||||
| search | 27 |
|
||||
| memory | 13 |
|
||||
| personal | 9 |
|
||||
| weather | 5 |
|
||||
| calendar | 3 |
|
||||
| attention | 3 |
|
||||
| list | 2 |
|
||||
| feeds | 2 |
|
||||
| tasks, money, self, habits | 1 each |
|
||||
|
||||
## What the badge shows that the wording did not
|
||||
|
||||
The two unfixed homelab turns are now direct evidence.
|
||||
|
||||
```text
|
||||
какая скорость у меня сейчас? -> weather
|
||||
хватает ли места под новые бэкапы? -> feeds
|
||||
```
|
||||
|
||||
Both are guessing sources claiming a turn about the box, exactly as the
|
||||
destination fixture predicted.
|
||||
|
||||
The badge also names a defect the wording hid. **Agenda questions are being
|
||||
claimed by the personal boundary and by Praxis, not by the calendar.**
|
||||
|
||||
```text
|
||||
во сколько у меня встреча? -> personal не нашла у тебя такой записи
|
||||
когда у меня встреча? -> attention у Praxis нет источников
|
||||
что у меня в понедельник? -> personal не нашла у тебя такой записи
|
||||
```
|
||||
|
||||
Calendar claimed 3 turns of the 6 that asked about the calendar. That is the
|
||||
same 3/6 the destination fixture scores and the same 3/6 every seed of the
|
||||
routing head scores. Three measurements agree. The cause is the one V-660 named. The possessive
|
||||
agenda rules claim these at stage 0 and name no destination, so the walk
|
||||
reaches `personal` and `attention` first.
|
||||
|
||||
This is the third independent confirmation that the possessive agenda rules
|
||||
should name the calendar. That call is still the owner's.
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
# CrisperWhisper 2.0 in Russian, measured
|
||||
|
||||
Date: 2026-08-09. Vikunja V-665.
|
||||
Corpus: `bond005/sberdevices_golos_10h_crowd`, test split, first 200 clips.
|
||||
Harness: `~/Programs/cw2-eval` on workpc, not in this repo.
|
||||
Runner: `./.venv/bin/python run_asr.py <arm>...` then `score.py`.
|
||||
|
||||
The model card benchmarks disfluency F1 in German and English. It never names
|
||||
Russian and publishes no per-language WER. So the measurement came before the
|
||||
wiring.
|
||||
|
||||
## The corpus
|
||||
|
||||
200 clips, 13.7 minutes, 1001 reference words. Median clip 3.91s, range 1.04s
|
||||
to 13.5s. Golos crowd is short crowd-sourced Russian spoken close to the
|
||||
microphone, which is the nearest public thing to someone talking to Maven. The
|
||||
alternatives are read speech, which flatters every model equally.
|
||||
|
||||
Two rows carry a null transcription and are skipped.
|
||||
|
||||
Scoring normalizes both sides: lowercase, `ё` to `е`, punctuation stripped, and
|
||||
digits expanded to Russian words through num2words. Without that last step a
|
||||
model is penalized for writing `60000` where the reference says
|
||||
`шестьдесят тысяч`. Thousands separators are joined before expansion, or
|
||||
`60 000` expands to `шестьдесят ноль`.
|
||||
|
||||
## Headline
|
||||
|
||||
| arm | WER | CER | exact | empty | RTF |
|
||||
|---|---|---|---|---|---|
|
||||
| cw2-turbo-intended | **10.4%** | 3.4% | 65.5% | 0 | 0.065 |
|
||||
| cw2-turbo-verbatim | 10.8% | **3.1%** | **66.5%** | 0 | 0.065 |
|
||||
| whisper-turbo | 11.8% | 4.1% | 64.0% | 0 | 0.031 |
|
||||
| cw2-large-intended | 12.3% | 3.8% | 63.5% | 0 | 0.107 |
|
||||
| whisper-small | 27.5% | 9.8% | 35.0% | 0 | 0.026 |
|
||||
|
||||
`whisper-small` is the floor, because `ggml-small.bin` is what mavsttd loads on
|
||||
homesrv today. CW2 turbo beats it by 17 points of WER and takes exact matches
|
||||
from 35.0% to 65.5%.
|
||||
|
||||
Two results are worth naming beyond the winner. CW2 turbo beats its own base
|
||||
model, whisper-large-v3-turbo, by 1.4 points. And it beats CW2 large by 1.9
|
||||
points, which inverts what the card implies by calling turbo a degraded draft.
|
||||
No arm returned an empty transcript.
|
||||
|
||||
## Intended and verbatim are closer than the mode names suggest
|
||||
|
||||
The two modes disagree on 70 of the 200 clips before normalization and on 29
|
||||
after it. So the raw difference is mostly casing and punctuation, which
|
||||
normalization removes and which Maven does not read either.
|
||||
|
||||
Verbatim scores worse on WER and better on CER and exact matches. The reason is
|
||||
script, not disfluency:
|
||||
|
||||
```text
|
||||
ref: футбольный матч челси брайтон
|
||||
int: Футбольный матч Chelsea-Брайтон.
|
||||
ver: Футбольный матч Челси Брайтон.
|
||||
```
|
||||
|
||||
Intended writes foreign entity names in Latin script and verbatim
|
||||
transliterates them. Golos references are Cyrillic throughout, so verbatim
|
||||
collects the exact matches. That is a property of this corpus rather than a
|
||||
quality difference.
|
||||
|
||||
**This corpus cannot settle the mode choice.** Golos crowd is clean short
|
||||
commands with almost no disfluency. The two modes have nothing to disagree
|
||||
about here. They separate on spontaneous speech with fillers, restarts and
|
||||
repairs, which is what the owner speaks. Intended stays the choice for the
|
||||
reason it was always the choice. Maven wants what was meant, not every stumble
|
||||
on the way there.
|
||||
|
||||
The Latin-script habit is the one finding here that touches routing. The
|
||||
routing heads were trained on Cyrillic utterances, so an entity name arriving
|
||||
in Latin script is out of distribution for them. Nothing measures that yet.
|
||||
|
||||
## The runtime is workpc, because whisper.cpp cannot load CW2
|
||||
|
||||
`num_languages()` in `deps/whisper.cpp/src/whisper.cpp` derives the language
|
||||
count from the vocabulary size:
|
||||
|
||||
```cpp
|
||||
return n_vocab - 51765 - (is_multilingual() ? 1 : 0);
|
||||
```
|
||||
|
||||
CW2 carries 31 extra tokens, so `n_vocab` is 51897 and this yields 131
|
||||
languages. The derived `dt` offset becomes 33 and shifts seven special token
|
||||
ids, including `token_beg` and `token_transcribe`. The architecture is
|
||||
otherwise byte-identical to whisper-large-v3-turbo, and the new tokens sit
|
||||
above every whisper special id.
|
||||
|
||||
So loading CW2 in whisper.cpp is a patch to a vendored dependency, not a port.
|
||||
It was not taken, because STT is moving to workpc anyway under V-486. CW2 turbo
|
||||
becomes the preferred remote and `ggml-small.bin` on homesrv stays the floor,
|
||||
which is the shape `modelSeam` already uses for routing and replies. The 27.5%
|
||||
floor is what a turn falls back to when the workstation is down, and this table
|
||||
is what that costs.
|
||||
|
||||
## License
|
||||
|
||||
Standard CW2 weights carry `nyra-health-non-commercial-research`. The Pro
|
||||
variants are commercial-license only. Maven is personal and self-hosted, so the
|
||||
standard weights are usable and the Pro ones are not free to take.
|
||||
|
||||
## What is not measured
|
||||
|
||||
Disfluent spontaneous speech, which is the whole reason to prefer Intended.
|
||||
Long-form audio beyond 13.5s. Far-field or noisy microphones. English, which
|
||||
Maven also speaks. The ONNX turbo export, which was never run, since the
|
||||
transformers path already meets the latency budget at RTF 0.065.
|
||||
@@ -47,6 +47,11 @@ type PendingQuestion struct {
|
||||
// charging it a retry is the V-554 shape. See CanResume for why it is
|
||||
// counted at all.
|
||||
Suspends int
|
||||
// Rides counts every turn this question has ridden out on the end of
|
||||
// someone else's reply, over the whole life of the request. Unlike Suspends
|
||||
// it is never reset and never re-based, which is the only property that
|
||||
// matters about it (V-663).
|
||||
Rides int
|
||||
}
|
||||
|
||||
// MaxSuspends — how many times one question may step aside and come back before
|
||||
@@ -63,10 +68,52 @@ type PendingQuestion struct {
|
||||
// is that he has moved on and has not said so.
|
||||
const MaxSuspends = 3
|
||||
|
||||
// MaxRides — how many turns one question may ride out on the end of an
|
||||
// unrelated reply, counted over its whole life (V-663).
|
||||
//
|
||||
// MaxSuspends did not move the measurement it was written for. Twenty-six of
|
||||
// 140 turns carried a tail before it landed and twenty-six carried one after.
|
||||
// Every bound on this question is rearmed by something ordinary:
|
||||
//
|
||||
// - The TTL is an inactivity timer, and both noteSuspended and reaskOrGiveUp
|
||||
// restart it, so it cannot arrive while he keeps talking.
|
||||
// - Suspends is zeroed by any turn that reads as an answer, which is where
|
||||
// "спасибо" and "привет" land. It resets before anything is known to have
|
||||
// been filled.
|
||||
// - askRemainingGap builds a fresh question for the second gap, so a reminder
|
||||
// with two gaps gets a new allowance halfway through.
|
||||
//
|
||||
// So Suspends only bites on four strictly consecutive side queries with nothing
|
||||
// chat-like between them, which is not the shape real conversation has. Rides is
|
||||
// the same idea with the resets taken out: set once, incremented, carried
|
||||
// across a re-park, and read by nothing that could lower it.
|
||||
//
|
||||
// The shape it is aimed at is measured, not imagined. In the 2026-08-08 run one
|
||||
// question about a reminder's day rode turns 7 to 13 and ended only because
|
||||
// turn 14 was a new request. Three asides, then two turns that read as failed
|
||||
// answers, then two more asides. The asides spend no attempt and the answers
|
||||
// reset Suspends, so the two bounds take turns being rearmed by the other's
|
||||
// traffic.
|
||||
//
|
||||
// Four, not three. It has to be looser than MaxSuspends or that bound is dead
|
||||
// code, because Rides is never lower than Suspends and would always fire first.
|
||||
//
|
||||
// Do not read this as a fix for the whole ride. It ends the measured one a turn
|
||||
// early and no more. Most of that ride's length is attempts, spent by turns
|
||||
// like "спасибо" and "привет" being read as failed answers to a question about
|
||||
// a day. That is a defect in classifyTurnRole and not in any bound here.
|
||||
const MaxRides = 4
|
||||
|
||||
// CanResume reports whether this question may step aside once more. False ⇒ the
|
||||
// caller lets the request go and says so; it must never simply stop resuming,
|
||||
// because a question dropped in silence reads as one that was answered.
|
||||
func (q *PendingQuestion) CanResume() bool { return q.Suspends < MaxSuspends }
|
||||
//
|
||||
// Two bounds, and they answer different questions. Suspends asks whether he has
|
||||
// walked away from this exchange in the last few turns. Rides asks whether this
|
||||
// question has been riding long enough that the answer is no regardless.
|
||||
func (q *PendingQuestion) CanResume() bool {
|
||||
return q.Suspends < MaxSuspends && q.Rides < MaxRides
|
||||
}
|
||||
|
||||
// Action reads the parked question as the typed action it is assembling
|
||||
// (pending.go). Derived rather than stored: the question's fields stay the one
|
||||
|
||||
@@ -119,6 +119,11 @@ func PartsOfDay() []string { return words("parts_of_day") }
|
||||
// ReminderVerbs returns the imperatives that open a reminder.
|
||||
func ReminderVerbs() []string { return words("reminder_verbs") }
|
||||
|
||||
// Pleasantries returns the whole utterances that greet, thank or say goodbye.
|
||||
// Whole utterances and not tokens: see the set's own note for why the tokens
|
||||
// are unsafe alone.
|
||||
func Pleasantries() []string { return words("pleasantries") }
|
||||
|
||||
// TaskDoneWords returns the words that finish a task, and TaskDropWords the
|
||||
// words that abandon one. Two sets rather than one with a value, because the
|
||||
// store records which of the two happened and the caller has to say so.
|
||||
|
||||
@@ -176,6 +176,19 @@
|
||||
"morning", "afternoon", "evening", "night"
|
||||
]
|
||||
},
|
||||
"pleasantries": {
|
||||
"note": "Whole utterances that greet, thank or say goodbye. They ask for nothing and answer nothing, so a parked question must neither consume them as a failed answer nor be dropped by them (V-663). Matched as WHOLE utterances and never as tokens, because the tokens are not safe alone: \"вечер\" answers \"это утра или вечера?\" and \"нет\" answers a confirm. Anything that could fill a slot stays out. The control words (\"стоп\", \"отмена\") stay out too, because isCancel already owns them and they mean something stronger.",
|
||||
"words": [
|
||||
"привет", "приветик", "здравствуй", "здравствуйте",
|
||||
"доброе утро", "добрый день", "добрый вечер",
|
||||
"пока", "прощай", "до свидания", "спокойной ночи",
|
||||
"спасибо", "спасибо тебе", "большое спасибо", "благодарю",
|
||||
"извини", "извините", "прости", "простите",
|
||||
"hi", "hello", "hey", "bye", "goodbye",
|
||||
"good morning", "good evening", "good night",
|
||||
"thanks", "thank you", "thanks a lot", "sorry"
|
||||
]
|
||||
},
|
||||
"reminder_verbs": {
|
||||
"note": "The imperatives that mean \"remind me\", in the forms he speaks. The same kind of set as capture_verbs and decided the same way: it is her vocabulary, not a discovery about Russian (Vikunja #530). The alarm verbs joined them in V-627. \"разбуди меня в 6:30\" is a reminder that fires at the hour he gets up, and the set knew no form of it, so an alarm reached IntentReminder only by resembling one to the embedder.",
|
||||
"words": [
|
||||
|
||||
@@ -54,7 +54,11 @@ def turn(text):
|
||||
q = urllib.parse.parse_qs(urllib.parse.urlparse(loc).query)
|
||||
return {
|
||||
"reply": q.get("r", [""])[0],
|
||||
"source": q.get("src", q.get("source", [""]))[0],
|
||||
# "s", not "src". cmd/mavweb/chat.go writes the badge under that
|
||||
# name, and reading the wrong one cost both fortnight runs their
|
||||
# source column: every finding in those docs is inferred from the
|
||||
# reply wording instead.
|
||||
"source": q.get("s", [""])[0],
|
||||
"trace": q.get("t", [""])[0],
|
||||
"secs": dt,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user