Compare commits
173 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| dc7c72a3d7 | |||
| 766ca091a7 | |||
| c5317eb2b4 | |||
| 9190f897a3 | |||
| d29e7ba813 | |||
| f7e1187823 | |||
| ed48c59ba7 | |||
| b09967f9e6 | |||
| b4a3867479 | |||
| 88d07b5175 | |||
| c00e3003bf | |||
| c0f9834528 | |||
| ad5eb2d1cf | |||
| 5253123d99 | |||
| 2abf98dea6 | |||
| f5c71b87f3 | |||
| 5934110fa8 | |||
| 6e47a3d736 | |||
| 67a5eb3805 | |||
| 029449eefa | |||
| ac36216f5d | |||
| db17cfcc65 | |||
| 7d676eb941 | |||
| fe3a4e9514 | |||
| a906f2afad | |||
| a2031a31d1 | |||
| 9b8bdf73cc | |||
| 7ad3c9a408 | |||
| 84e1478823 | |||
| 9c8d0baffe | |||
| b0f5a16ec9 | |||
| 67563ed1f6 | |||
| f0f7ebc9b2 | |||
| d12de589a2 | |||
| 50cc17f33a | |||
| 73d13f1ea6 | |||
| 0ca5748699 | |||
| b43bb265b5 | |||
| b9a24334ea | |||
| c97aebf55a | |||
| 891136c65d | |||
| 41c7c13f42 | |||
| a324e8f624 | |||
| 51805e7f35 | |||
| 533f0acda8 | |||
| 4f59ba78c6 | |||
| d0afd9d4f6 | |||
| 6b67e6f3c2 | |||
| 742b2ad1d7 | |||
| b300ac5c70 | |||
| 13e5170e9e | |||
| 0b90952e55 | |||
| aa8f5b2ee2 | |||
| d7cdcb63bd | |||
| ddb658ffbb | |||
| c7dadc97d9 | |||
| c9d88c152e | |||
| 1890ff5d5d | |||
| 0110e9bc8c | |||
| 50ca8c8b5a | |||
| de09471421 | |||
| d65c16a567 | |||
| 062d4252ef | |||
| 2c27e2ce1f | |||
| ccc5cba2a3 | |||
| 89d83c0b11 | |||
| a97f554802 | |||
| f4de2fc5e1 | |||
| eef5d4da4f | |||
| 09f1696fce | |||
| 80f7322294 | |||
| fa5aebfbe4 | |||
| 59cec63da1 | |||
| 02e8786695 | |||
| 0272dc9d89 | |||
| 2ad7635501 | |||
| 9949b309b1 | |||
| a788ca3915 | |||
| 62d47d28ac | |||
| e9ff2c4912 | |||
| 3dbf67f8f9 | |||
| 84ba217892 | |||
| f179ae2fde | |||
| d00929ac0b | |||
| b6f47fbeb6 | |||
| 2e9b9ec1cf | |||
| bfb57c3148 | |||
| d1f6f6355f | |||
| 92ecb691de | |||
| 4282f6b9a9 | |||
| 7bb9f9be06 | |||
| 1e47eaca5a | |||
| 892330eb84 | |||
| 9a3bcd7c46 | |||
| 98ee701e03 | |||
| 04c1088088 | |||
| 07c191d8b8 | |||
| c668310b3e | |||
| 1bd2acdc2a | |||
| 15e5dd8eaa | |||
| 10cf6f525c | |||
| e2210f6844 | |||
| 214a4032cf | |||
| dc70a5a7ab | |||
| 06aded6ab0 | |||
| d2be98ee2a | |||
| 62d320f93a | |||
| 796e6af3cf | |||
| 74a70880a8 | |||
| a40bc559d5 | |||
| 1db0fcfcd0 | |||
| 4ca68d2f3f | |||
| 5a1d465db5 | |||
| 43838445ab | |||
| 11831c6ace | |||
| 93c1a41d4a | |||
| bce5ed210c | |||
| c31f0d1001 | |||
| 34521c30b8 | |||
| 1d48755d12 | |||
| f6d5a2a7a4 | |||
| 751c2a705f | |||
| 5d5b0cfd49 | |||
| bd16ca69e5 | |||
| 0ed386eca6 | |||
| 1c4eab2107 | |||
| 0914e0a3d5 | |||
| c860808528 | |||
| f7442c3aea | |||
| 75b067ac51 | |||
| 424d1b3446 | |||
| c47886c2bc | |||
| f6236da760 | |||
| 54dc43516b | |||
| fa51a48958 | |||
| 5fd25d7ad7 | |||
| c22fc352fc | |||
| 215aa331c5 | |||
| 859bbf750f | |||
| ee3e6a9eaf | |||
| 8acb8a97c6 | |||
| 1f55207b58 | |||
| 4db109346a | |||
| 2f00593411 | |||
| 0b3b8d0a9e | |||
| fe0e654ab1 | |||
| a54ebac0cb | |||
| 784d688b44 | |||
| 4ba9a6f422 | |||
| 32687b3712 | |||
| db3e706bdc | |||
| 2f4257e194 | |||
| 7d8b0af99d | |||
| cfd38d53cf | |||
| a2835bbdf6 | |||
| c262c1e4c0 | |||
| a33ad82178 | |||
| af35ec3629 | |||
| 9145b83100 | |||
| 94eb92fb15 | |||
| 707c3e5040 | |||
| ace9fbbc06 | |||
| 3884db33e9 | |||
| dfb8d26b62 | |||
| bf99fd4192 | |||
| 6d9aa83b6e | |||
| f75072175d | |||
| 39d83a33e8 | |||
| 9d8fcf42f3 | |||
| 925ce223a0 | |||
| 9e2af3286b | |||
| d30618ecb7 | |||
| 17b47ce206 |
+11
-1
@@ -6,12 +6,16 @@
|
|||||||
/mavweb
|
/mavweb
|
||||||
/mavpoll
|
/mavpoll
|
||||||
/mavcaldav
|
/mavcaldav
|
||||||
|
/mavwaked
|
||||||
|
|
||||||
# Certs (private keys, don't commit)
|
# Certs (private keys, don't commit)
|
||||||
certs/
|
certs/
|
||||||
|
|
||||||
# Dependencies (fetch/build, not vendored)
|
# Dependencies (fetch/build, not vendored).
|
||||||
|
# Both forms on purpose: 'deps/' misses a symlink named deps, and agents working
|
||||||
|
# in a git worktree symlink these in from the main checkout.
|
||||||
deps/
|
deps/
|
||||||
|
deps
|
||||||
|
|
||||||
# ML models (large, downloaded separately) — specific dirs, not blanket,
|
# ML models (large, downloaded separately) — specific dirs, not blanket,
|
||||||
# because models/seeds/*.txt are small, tracked files the classifier needs.
|
# because models/seeds/*.txt are small, tracked files the classifier needs.
|
||||||
@@ -19,6 +23,9 @@ deps/
|
|||||||
/models/stt/
|
/models/stt/
|
||||||
/models/tts/
|
/models/tts/
|
||||||
/models/llm/
|
/models/llm/
|
||||||
|
# Symlink forms, same reason as deps above.
|
||||||
|
/models/embedder
|
||||||
|
/models/llm
|
||||||
|
|
||||||
# Runtime data
|
# Runtime data
|
||||||
*.db
|
*.db
|
||||||
@@ -36,3 +43,6 @@ opencode.json
|
|||||||
|
|
||||||
# Test coverage output
|
# Test coverage output
|
||||||
coverage.out
|
coverage.out
|
||||||
|
|
||||||
|
# Agent worktrees and local agent state
|
||||||
|
.claude/
|
||||||
|
|||||||
@@ -43,14 +43,17 @@ notes. Without it, the floor `HashEmbedder` is used — deterministic but weak
|
|||||||
(Russian recall rarely clears the confidence gate, many commands fall to
|
(Russian recall rarely clears the confidence gate, many commands fall to
|
||||||
"clarify").
|
"clarify").
|
||||||
|
|
||||||
**Download the embedder** (ONNX, ~90 MB):
|
**Download the embedder** (ONNX, ~120 MB):
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
make download-embedder
|
make download-embedder
|
||||||
```
|
```
|
||||||
|
|
||||||
This fetches `paraphrase-multilingual-MiniLM-L12-v2` (384-dim, 12-layer,
|
This fetches `multilingual-e5-small` (384-dim, 12-layer, Russian and English)
|
||||||
supports 50+ languages including Russian) to `models/embedder/`.
|
to `models/embedder/multilingual-e5-small/`. It is an asymmetric retrieval
|
||||||
|
model: the code puts `query: ` in front of a question and `passage: ` in front
|
||||||
|
of a stored note, which is how e5 was trained. The quantized file is the one
|
||||||
|
that is downloaded, deployed and measured.
|
||||||
|
|
||||||
**Also need ONNX Runtime** (`libonnxruntime.so`):
|
**Also need ONNX Runtime** (`libonnxruntime.so`):
|
||||||
|
|
||||||
@@ -64,8 +67,8 @@ sudo cp onnxruntime-linux-x64-1.15.1/lib/libonnxruntime.so* /usr/local/lib/
|
|||||||
```json
|
```json
|
||||||
"voice": {
|
"voice": {
|
||||||
"embedder": {
|
"embedder": {
|
||||||
"model_path": "models/embedder/model_quantized.onnx",
|
"model_path": "models/embedder/multilingual-e5-small/model_quantized.onnx",
|
||||||
"tokenizer_path": "models/embedder/tokenizer.json",
|
"tokenizer_path": "models/embedder/multilingual-e5-small/tokenizer.json",
|
||||||
"lib_path": "/usr/local/lib/libonnxruntime.so"
|
"lib_path": "/usr/local/lib/libonnxruntime.so"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,9 +7,20 @@ talking over unix sockets; one resident small model for routing + phrasing; whis
|
|||||||
Deploy target is a Ryzen laptop (homesrv) with Vulkan offload to the Vega iGPU (`n_gpu_layers: 99`,
|
Deploy target is a Ryzen laptop (homesrv) with Vulkan offload to the Vega iGPU (`n_gpu_layers: 99`,
|
||||||
compose passes `/dev/dri` + the render gid) — the resident model stays ≤1.7B either way.
|
compose passes `/dev/dri` + the render gid) — the resident model stays ≤1.7B either way.
|
||||||
|
|
||||||
**Resident model:** currently **Qwen3.5-0.8B** (`Q4_K_M`), the smallest checkpoint in the gguf
|
**Resident model:** currently **Qwen3-1.7B** (`UD-Q4_K_XL`), stock — not yet the CPT'd one.
|
||||||
library, picked for CPU/iGPU latency. The **target** is the locally CPT'd **Qwen3-1.7B**; that
|
It replaced Qwen3.5-0.8B on 2026-07-31 because it measured better on both fixtures we have:
|
||||||
training is still in flight (Vikunja #122), so no such gguf exists yet. Model files live in
|
67.5% vs 59.7% intent-only on the 77-case RU routing fixture, and 20/27 vs 11-17/27 on the
|
||||||
|
talk fixture. See `MODEL-BAKEOFF-31-07-2026.md`. It is a Thinking variant, so `n_ctx` is 4096
|
||||||
|
— reasoning tokens need the room, and 4096 is what the scores above were measured at.
|
||||||
|
|
||||||
|
The **target** is still the locally CPT'd **Qwen3-1.7B** (Vikunja #122, training in flight).
|
||||||
|
Stock already speaks good Russian; what it gets wrong is the persona — it writes `я рад`,
|
||||||
|
masculine, where Maven needs `рада`. That is what the CPT is for.
|
||||||
|
|
||||||
|
**Do not bother with sub-500M models.** LFM2.5-230M and 350M were measured on 2026-07-31 and
|
||||||
|
both are unusable in Russian: the 350M routes at 5.2% (worse than guessing) and answers
|
||||||
|
"столица Франции?" with the invented non-word "Сторзит"; the 230M replies to Russian in
|
||||||
|
Spanish. Their strong published IFEval/BFCL numbers are English-only. Model files live in
|
||||||
`/mnt/hdd1/llms`, bind-mounted to `/opt/maven/models/llm` — which **shadows** the repo's
|
`/mnt/hdd1/llms`, bind-mounted to `/opt/maven/models/llm` — which **shadows** the repo's
|
||||||
`models/llm/`, so the LFM2.5 gguf sitting there is not loaded by anything. Swapping the resident
|
`models/llm/`, so the LFM2.5 gguf sitting there is not loaded by anything. Swapping the resident
|
||||||
model is a one-line change to `phraser.model_path` in `deploy/mavend.json`.
|
model is a one-line change to `phraser.model_path` in `deploy/mavend.json`.
|
||||||
@@ -58,19 +69,41 @@ protocol; the config in `deploy/mavend.json` (with `${VAR}` env expansion from g
|
|||||||
|
|
||||||
## Routing — read this before touching the router
|
## Routing — read this before touching the router
|
||||||
|
|
||||||
`internal/router/` has TWO layered engines and the committed default is an **interim
|
`internal/router/` has TWO layered engines. **The LLM router is now the default and it is
|
||||||
stopgap, not the intended design** (see memory `routing-architecture-target`):
|
on in deploy** — this section used to say it was wired `nil`, which stopped being true on
|
||||||
|
2026-07-31.
|
||||||
|
|
||||||
- **Target (REARCH.md):** LLM-as-router. One resident Qwen3-1.7B (`llmrouter.go`) emits
|
- **LLM router (the intended design, REARCH.md):** the resident Qwen3-1.7B (`llmrouter.go`)
|
||||||
GBNF-constrained structured JSON, and the SAME model phrases replies. Embedder is demoted
|
emits GBNF-constrained structured JSON, and the SAME model phrases replies. Embedder is
|
||||||
from a routing gate to a RAG hint.
|
demoted from a routing gate to a RAG hint. Wired at `voice.go:214` via
|
||||||
- **Current stopgap:** `llmrouter` is wired `nil` (around `voice.go`), so the
|
`pickLLMRouter(cfg.Voice.UseLLMRouter(), llmClient)`; the flag is `voice.llm_router`
|
||||||
`classifier.go` + `embedder.go` nearest-neighbour cascade actually runs. It routes by
|
(`config.go`), `DefaultLLMRouter` is **on**, and `deploy/mavend.json` sets it `true`.
|
||||||
similarity to frozen seed phrases — the known cause of weak RU query handling.
|
- **Classifier cascade (the failure floor, not dead code):** `classifier.go` +
|
||||||
|
`embedder.go` nearest-neighbour over frozen seed phrases. It runs when the LLM router is
|
||||||
|
off, when there is no llama-server to talk to (`pickLLMRouter` logs that and degrades),
|
||||||
|
and on any per-turn LLM error. Do not delete it — routing by seed similarity is the known
|
||||||
|
cause of weak RU query handling, but a turn must never break on the model.
|
||||||
|
|
||||||
Cascade order: `stage0.go` exact-match fast-path → LLM router (when non-nil) → classifier
|
Cascade order: `stage0.go` exact-match fast-path → LLM router (when non-nil) → classifier
|
||||||
fallback. Any LLM error falls through to the classifier so a turn never breaks on the model.
|
fallback. Any LLM error falls through to the classifier so a turn never breaks on the model.
|
||||||
|
|
||||||
|
Measured on the 77-case RU fixture (`MODEL-BAKEOFF-31-07-2026.md`): the classifier scores
|
||||||
|
36.8% full accuracy at p50 31ms; Qwen3-1.7B scores 67.5% intent-only / 72.7% through the
|
||||||
|
cascade at p50 ≈2.7s. Accuracy roughly doubled, latency is ~90× worse, and that trade was
|
||||||
|
accepted deliberately. `Confidence: 1.0` used to be hardcoded in `llmrouter.go`, so the LLM
|
||||||
|
path could never ask for clarification (6/6 refusal cases missed on the fixture) — Vikunja
|
||||||
|
#359. Fixed 31-07-2026 with structural signal (single-token utterance, keyless fact, act with
|
||||||
|
no allowlisted fn) feeding the same stage-3 gate the classifier path already had — see
|
||||||
|
`gateLLMDecision` in `router.go`. Note the second half of that bug: the LLM branch never
|
||||||
|
consulted `r.threshold` at all, so a correct low confidence would have been discarded anyway.
|
||||||
|
|
||||||
|
Re-measured on the fixture after the fix: **missed clarify 6/6 → 1**, at the cost of 3 false
|
||||||
|
clarifies and 2.6pt of full accuracy (72.7% → 70.1%, intent-only 67.5% → 74.0%). Two of the
|
||||||
|
three false clarifies are acts the model mis-routed and the gate caught — asking beats wrongly
|
||||||
|
executing, so the fixture and the daemon disagree about what is correct there. The third,
|
||||||
|
`"поужинал"`, is a real defect: **the single-token rule is an English intuition and does not
|
||||||
|
transfer to Russian**, where one word is routinely a whole sentence. Narrow or drop it.
|
||||||
|
|
||||||
## LLM output contract
|
## LLM output contract
|
||||||
|
|
||||||
All phrasing paths emit `{"response":"...","mood":"..."}` (parsed in `replier_llm.go` and
|
All phrasing paths emit `{"response":"...","mood":"..."}` (parsed in `replier_llm.go` and
|
||||||
@@ -82,8 +115,26 @@ workspace enforces that the Go and relabelling prompts remain identical.
|
|||||||
|
|
||||||
## Non-goals (hard constraints)
|
## Non-goals (hard constraints)
|
||||||
|
|
||||||
Never phones home. Not a nag, not autonomous. Maven's persona is **feminine** — Russian
|
Not a nag, not autonomous. Maven's persona is **feminine** — Russian
|
||||||
self-reference must use feminine forms (the user is male; see memory `maven-persona-gender`).
|
self-reference must use feminine forms — `рада`, not `рад`; `поняла`, not `понял`. The owner
|
||||||
|
is male and is addressed informally: "ты", singular, never "вы"/"ваш" and never "он"/"его"
|
||||||
|
(she talks TO him, not about him). Pet names ("милый", "дорогой") are forbidden; his name
|
||||||
|
("Ками") is not. The eval enforces this: `CheckAddress`, `CheckFeminine` and `CheckCringe` in
|
||||||
|
`internal/phraser/eval/checks.go`, scored by `make eval-phrasing`.
|
||||||
|
|
||||||
|
**"Never phones home" is DEPRECATED** (owner's call, 2026-07-31). It used to be a hard
|
||||||
|
constraint and it is not one any more: a 0.8B — and a 1.7B — does not know enough to answer
|
||||||
|
world questions, so she needs to read external sources. What replaces it:
|
||||||
|
|
||||||
|
- **No telemetry, no cloud model, no third-party account.** That part never changes. Nothing
|
||||||
|
about Maven is reported to anyone, and inference stays on the box.
|
||||||
|
- **Local sources first.** Kiwix ZIMs on homesrv (Wikipedia, ifixit) before anything on the
|
||||||
|
network. Reading beats recalling for a small model, and a local read costs nothing.
|
||||||
|
- **External search is allowed and off unless configured**, like the weather and telegram
|
||||||
|
capabilities.
|
||||||
|
- **His notes and facts are never search input.** Looking up why the sky is blue and sending
|
||||||
|
his stored personal notes to an upstream engine are different acts. Only the utterance goes
|
||||||
|
out, never the persona block, history, or matched notes.
|
||||||
|
|
||||||
## Web UI conventions
|
## Web UI conventions
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,8 @@
|
|||||||
|
|
||||||
**Maven** — self-hosted personal assistant. Manages your day, acts on your
|
**Maven** — self-hosted personal assistant. Manages your day, acts on your
|
||||||
homelab. One daemon on homesrv (always-on, not the workstation), multiple
|
homelab. One daemon on homesrv (always-on, not the workstation), multiple
|
||||||
client surfaces. All local, never phones home.
|
client surfaces. Inference and data stay on the box; she may READ external
|
||||||
|
sources (see Non-goals — "never phones home" is deprecated).
|
||||||
|
|
||||||
Primary name is "Maven", with feminine-gendered Russian self-reference
|
Primary name is "Maven", with feminine-gendered Russian self-reference
|
||||||
("она", "меня", "помогла"). Clients may choose their own UI label. Consistent
|
("она", "меня", "помогла"). Clients may choose their own UI label. Consistent
|
||||||
@@ -35,8 +36,13 @@ Inside boundary — the ones that actually constrain the build:
|
|||||||
she records. A confident wrong fact is worse than a known gap.
|
she records. A confident wrong fact is worse than a known gap.
|
||||||
- **Not a nag** — she'd rather miss a nudge than be mutable. Shuts up when
|
- **Not a nag** — she'd rather miss a nudge than be mutable. Shuts up when
|
||||||
uncertain. Load-bearing.
|
uncertain. Load-bearing.
|
||||||
- **Not a stranger** — runs on your stuff, your model, your data. Never
|
- **Not a stranger** — runs on your stuff, your model, your data. No
|
||||||
phones home.
|
telemetry, no cloud model, no third-party account. She may READ external
|
||||||
|
sources to answer world questions (Kiwix first, then optional search); she
|
||||||
|
never reports anything about you to anyone, and your notes and facts are
|
||||||
|
never used as search input. **"Never phones home" as an absolute is
|
||||||
|
deprecated** — owner's call, 2026-07-31: a small model does not know enough
|
||||||
|
to be useful without reading.
|
||||||
- **Not a relationship** — mom-tone is a function that makes nudges land, not
|
- **Not a relationship** — mom-tone is a function that makes nudges land, not
|
||||||
emotional company. Names the drift a warm small model falls into.
|
emotional company. Names the drift a warm small model falls into.
|
||||||
|
|
||||||
@@ -458,7 +464,7 @@ decides *insistence*. Both are needed.
|
|||||||
|
|
||||||
sev ≤ 2 drops on away, sev ≥ 3 holds: a missed water nudge is noise, a missed
|
sev ≤ 2 drops on away, sev ≥ 3 holds: a missed water nudge is noise, a missed
|
||||||
backup failure isn't. Away-channels (ntfy/telegram) leave the box — the one
|
backup failure isn't. Away-channels (ntfy/telegram) leave the box — the one
|
||||||
path that crosses "never phones home," through your own relay. **Minimal
|
path that leaves the box for a person to see, through your own relay. **Minimal
|
||||||
body** — "disk low on homesrv," not detail; don't make notifications a
|
body** — "disk low on homesrv," not detail; don't make notifications a
|
||||||
shoulder-surf exfil surface.
|
shoulder-surf exfil surface.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,223 @@
|
|||||||
|
# Resident model bake-off — 31-07-2026
|
||||||
|
|
||||||
|
**Outcome: the resident model is stock Qwen3-1.7B** (`UD-Q4_K_XL`). Two sweeps ran this
|
||||||
|
evening and the second one changed the answer — read to the end before acting on any table
|
||||||
|
here. [Second sweep](#second-sweep-same-evening--five-models-and-a-resident-model-change)
|
||||||
|
is the one that holds.
|
||||||
|
|
||||||
|
## First sweep — LFM2.5-1.2B vs Qwen3.5-0.8B
|
||||||
|
|
||||||
|
**Verdict, scoped to this pair: keep Qwen3.5-0.8B over LFM2.5-1.2B.** LFM2.5-1.2B is worse
|
||||||
|
at routing (52.6% vs 60.5% intent accuracy), and the loss is almost entirely Russian
|
||||||
|
(18/61 vs 22/61 RU, while EN is a wash). It is also 2.4× slower. The Thinking variant is
|
||||||
|
far worse again. This verdict still stands as written — it rejects LFM2.5-1.2B. It is
|
||||||
|
**not** a recommendation to keep 0.8B as the resident model; the second sweep replaced it
|
||||||
|
with Qwen3-1.7B.
|
||||||
|
|
||||||
|
Settles Vikunja **#278 / #250**.
|
||||||
|
|
||||||
|
- Same fixture and scorer as `ROUTING-EVAL-31-07-2026.md`: `internal/router/eval/`
|
||||||
|
(`ru_routing_v1.json`, 76 held-out cases).
|
||||||
|
- Reproduce: `MAVEN_LLM_URL=http://127.0.0.1:<port> make eval-router`
|
||||||
|
(`TestLLMRouterBaseline`). (This line used to say there is no `make eval-models` target.
|
||||||
|
There is one now — start a server with the gguf you want, then
|
||||||
|
`make eval-models MAVEN_LLM_URL=http://127.0.0.1:<port>`. It runs only the LLM test, since
|
||||||
|
the classifier baselines do not depend on the model.)
|
||||||
|
- All three models served by the same `llama-server` flags — `-c 2048 -ngl 99 -t 6`, only
|
||||||
|
`-m` and `--port` differ. One server at a time on an otherwise idle box, so latencies are
|
||||||
|
real and not contention.
|
||||||
|
- Measured on top of the router prompt fix (`origin/overnight/router-prompt` merged in), so
|
||||||
|
the Qwen column is directly comparable to the numbers already recorded.
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
`llm-only` — the model alone. This is the column that measures the model.
|
||||||
|
|
||||||
|
| | Qwen3.5-0.8B | LFM2.5-1.2B Instruct | LFM2.5-1.2B Thinking |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **intent-only accuracy** | **60.5%** | 52.6% | 36.8% |
|
||||||
|
| full accuracy (intent+slots+gate) | **36.8%** | 32.9% | 21.1% |
|
||||||
|
| **RU** | **22/61** | 18/61 | 10/61 |
|
||||||
|
| EN | 6/15 | **7/15** | 6/15 |
|
||||||
|
| route errors | 0 | 0 | 0 |
|
||||||
|
| **p50 / p95 latency** | **1.05s / 1.71s** | 2.47s / 3.62s | 2.42s / 3.24s |
|
||||||
|
| missed clarify | 6 / 6 | 6 / 6 | 6 / 6 |
|
||||||
|
|
||||||
|
`cascade+llm` — stage-0 → model → classifier floor, what #320 would actually ship. Same
|
||||||
|
ordering.
|
||||||
|
|
||||||
|
| | Qwen3.5-0.8B | LFM2.5-1.2B Instruct | LFM2.5-1.2B Thinking |
|
||||||
|
|---|---|---|---|
|
||||||
|
| intent-only accuracy | **61.8%** | 55.3% | 38.2% |
|
||||||
|
| full accuracy | **46.1%** | 42.1% | 30.3% |
|
||||||
|
| RU / EN | **27/61** / 8/15 | 23/61 / **9/15** | 15/61 / 8/15 |
|
||||||
|
| route errors | 0 | 0 | 0 |
|
||||||
|
| p50 / p95 latency | **1.28s / 1.94s** | 2.18s / 2.72s | 2.27s / 3.19s |
|
||||||
|
|
||||||
|
Full logs: the three runs are archived in the session scratchpad
|
||||||
|
(`qwen08.txt`, `lfm-instruct.txt`, `lfm-thinking.txt`).
|
||||||
|
|
||||||
|
## Russian-specific failures — the owner's worry is confirmed
|
||||||
|
|
||||||
|
LFM2.5's Russian loss is not spread out. It has one large, specific failure: **it hears
|
||||||
|
almost any Russian imperative or short phrase as `reminder`.**
|
||||||
|
|
||||||
|
- `перезапусти докер` → reminder (want act)
|
||||||
|
- `включи вытяжку` → reminder (want act)
|
||||||
|
- `закрой жалюзи` → reminder (want act)
|
||||||
|
- `заметка: продлить домен в августе` → reminder (want note)
|
||||||
|
- `запиши что кран на кухне снова капает` → reminder (want note)
|
||||||
|
- `доброе утро` → reminder (want chat)
|
||||||
|
- `спасибо тебе` → reminder (want note/chat)
|
||||||
|
- `переходи в тихий режим` → reminder (want system)
|
||||||
|
|
||||||
|
That is `note→reminder ×4`, `act→reminder ×4`, `chat→reminder ×2` in one run. Qwen's
|
||||||
|
equivalent failure axis is `query→fact ×8`, which is a narrower and already-understood bug.
|
||||||
|
|
||||||
|
Two more Russian-side problems worth naming:
|
||||||
|
|
||||||
|
1. **Fact keys come back empty or wrong in Russian.** `воды попил наконец`, `поужинал`,
|
||||||
|
`поспал часов пять` and `отметь что я позавтракал овсянкой` all returned an empty key.
|
||||||
|
`сходил в душ` and `отдохнул минут двадцать` both returned `water`. Qwen does not do this.
|
||||||
|
2. **It leaked German.** `slept about seven hours` produced the fact key
|
||||||
|
`"7 Stunden geschlafen"`. Grammar-valid, semantically garbage — a sign the multilingual
|
||||||
|
mix is not anchored where Maven needs it.
|
||||||
|
|
||||||
|
The claimed tool-calling advantage did not show up here. `act` is the closest thing this
|
||||||
|
fixture has to a tool call, and LFM2.5 got it wrong more often than Qwen, mostly by calling
|
||||||
|
it a reminder. It also produced no `fn` slot on any act, same as Qwen.
|
||||||
|
|
||||||
|
## The Thinking variant
|
||||||
|
|
||||||
|
Not viable. 36.8% intent accuracy, 10/61 Russian, and no latency saving over Instruct — the
|
||||||
|
thinking trace costs time without buying accuracy on a short enum classification. With the
|
||||||
|
`enable_thinking=false` diagnostic it collapsed further to 28.9% with 2 route errors
|
||||||
|
(`query→reminder ×12`). Do not pursue.
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- Nothing crashed, nothing ignored the GBNF grammar, and no model produced unparseable JSON
|
||||||
|
in the shippable configurations. Zero route errors for both Instruct and Thinking in
|
||||||
|
`llm-only` and `cascade+llm`. The problem with LFM2.5 is what it decides, not whether it
|
||||||
|
can emit the contract.
|
||||||
|
- The `6 / 6` missed clarify is unchanged across all three models. No model fixes the missing
|
||||||
|
refusal lane — that is `Confidence: 1.0` hardcoded in `llmrouter.go` (Vikunja #359), not a
|
||||||
|
model property.
|
||||||
|
- The report labels every configuration `(0.8B)`; that string is hardcoded in the test, not a
|
||||||
|
reflection of which gguf was loaded. Model identity was confirmed per run via `/v1/models`.
|
||||||
|
- No Go code was changed for this measurement, and no bug was found that needed one.
|
||||||
|
|
||||||
|
## What this does not settle
|
||||||
|
|
||||||
|
Routing only. LFM2.5 might still phrase better, and phrasing is the resident model's other
|
||||||
|
job — that needs its own fixture. But routing is the load-bearing path and Maven is
|
||||||
|
Russian-first, so on the evidence here the switch is not worth making.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
# Second sweep, same evening — five models, and a resident-model change
|
||||||
|
|
||||||
|
The sections above compared LFM2.5-1.2B against Qwen3.5-0.8B on routing and concluded
|
||||||
|
"the switch is not worth making". That still holds. This sweep asked a different
|
||||||
|
question — whether a *smaller* model could work, since LFM2.5's published
|
||||||
|
instruction-following scores beat Qwen3.5-0.8B badly — and answered it, plus found a
|
||||||
|
better resident model by accident.
|
||||||
|
|
||||||
|
**Outcome: the resident model is now stock Qwen3-1.7B.** Sub-500M is a dead end.
|
||||||
|
|
||||||
|
## Routing — 77 Russian cases, one run each
|
||||||
|
|
||||||
|
| model | on disk | llm-only (full) | llm-only (intent) | cascade + fallback |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| LFM2.5-230M-Q8_0 | 246 MB | 23.4% | 33.8% | 36.4% |
|
||||||
|
| LFM2.5-350M-Q8_0 | 379 MB | 2.6% | **5.2%** | 20.8% |
|
||||||
|
| Qwen3.5-0.8B-Q4_K_M | 527 MB | 36.4% | 59.7% | 61.0% |
|
||||||
|
| Qwen3.5-2B-UD-Q4_K_XL | 1.34 GB | 42.9% | 62.3% | 63.6% |
|
||||||
|
| **Qwen3-1.7B-UD-Q4_K_XL (stock)** | 1.13 GB | **44.2%** | **67.5%** | **72.7%** |
|
||||||
|
|
||||||
|
Qwen3-1.7B wins every column, including against a model 20% larger than it.
|
||||||
|
|
||||||
|
## Talk fixture — 27 cases, three runs each, idle box
|
||||||
|
|
||||||
|
| | Qwen3.5-0.8B | Qwen3-1.7B stock |
|
||||||
|
|---|---|---|
|
||||||
|
| composite | 13, 11, 8 | **20, 21, 18** |
|
||||||
|
| address | 21, 18, 18 | **26, 25, 23** |
|
||||||
|
| feminine | 27, 25, 26 | 26, 27, 26 |
|
||||||
|
| lang | 27, 27, 26 | 26, 27, 27 |
|
||||||
|
| ontopic | 16, 19, 19 | **22, 23, 23** |
|
||||||
|
| canned fallbacks | 8, 5, 6 | **0, 2, 0** |
|
||||||
|
|
||||||
|
This also fills the row `TALK-EVAL-31-07-2026.md` had to void for contamination:
|
||||||
|
**600ch/1024tok on Qwen3.5-0.8B scores 13, 11, 8.**
|
||||||
|
|
||||||
|
`address` is the headline. It sat at 18-22 of 27 on the 0.8B no matter how the prompt
|
||||||
|
was worded — the prompt explicitly forbids "вы" and the model writes `вашей`,
|
||||||
|
`подождите`, `делаете` anyway. That was read as "prompting is out of levers", and it
|
||||||
|
was really "0.8B is out of capacity". The 1.7B mostly holds the constraint.
|
||||||
|
|
||||||
|
The fallback column matters too: 5-8 of 27 turns on the 0.8B end in a hardcoded
|
||||||
|
`"не знаю."`, meaning it failed to emit parseable JSON about a quarter of the time.
|
||||||
|
The 1.7B does that 0-2 times.
|
||||||
|
|
||||||
|
## Latency — the long tail is not the Thinking block
|
||||||
|
|
||||||
|
| | p50 | p95 |
|
||||||
|
|---|---|---|
|
||||||
|
| Qwen3.5-0.8B | 2.4s, 2.9s, 2.0s | 17.4s, 17.6s, 17.4s |
|
||||||
|
| Qwen3-1.7B stock | 2.7s, 2.6s, 2.8s | 16.4s, 6.6s, 3.9s |
|
||||||
|
|
||||||
|
p50 is flat across a 2× size difference. The first instinct on seeing the 1.7B's
|
||||||
|
16s p95 was "that is the reasoning trace, cap it" — wrong. The 0.8B's p95 is a
|
||||||
|
consistent 17s and the 1.7B beat it in two of three runs. The tail is shared and
|
||||||
|
lives somewhere else. Do not spend time on `/no_think` on this evidence.
|
||||||
|
|
||||||
|
## Sub-500M: not close, and the benchmarks say otherwise for a reason
|
||||||
|
|
||||||
|
LFM2.5-350M publishes IFEval 76.96 against Qwen3.5-0.8B's 59.94, and BFCLv3 44.11
|
||||||
|
against 35.08 — better at instruction-following and structured output, at 2/3 the
|
||||||
|
size. Those numbers are real and they are **English**. Every benchmark in that
|
||||||
|
table except Multi-IF is English-only.
|
||||||
|
|
||||||
|
In Russian, with a 300-token budget and temperature 0:
|
||||||
|
|
||||||
|
- **350M**, «Столица Франции? Ответь кратко.» → *«Сторзит в Париже.»* — `Сторзит` is
|
||||||
|
not a word; it is invented morphology.
|
||||||
|
- **350M**, asked to read back a reminder → a fortune cookie about being attentive
|
||||||
|
and confident. No reminder in it.
|
||||||
|
- **230M**, «Привет, как дела?» → answered **in Spanish**.
|
||||||
|
|
||||||
|
The 230M beating the 350M six-fold on routing (33.8% vs 5.2%) is the other tell:
|
||||||
|
when the larger sibling collapses like that it is format compliance failing, not
|
||||||
|
reasoning.
|
||||||
|
|
||||||
|
This is a pretraining gap, not a fine-tuning gap. Teaching Russian to a 350M from
|
||||||
|
near-zero is not an afternoon on a Colab, which was the premise worth checking.
|
||||||
|
|
||||||
|
## Why this vindicates the 1.7B CPT
|
||||||
|
|
||||||
|
Stock Qwen3-1.7B, untrained and unprompted, answers all three probes in fluent
|
||||||
|
correct Russian. What it gets wrong is the persona: *«Привет! Я рад, что ты здесь»*
|
||||||
|
— `рад` is masculine and Maven needs `рада`. That is the right kind of remaining
|
||||||
|
problem, and it is exactly what the CPT (Vikunja #122) is for.
|
||||||
|
|
||||||
|
The 1.7B was the correct model choice. What was wrong was treating it as a
|
||||||
|
**blocker**: stock already beats what was deployed, so it ships now and gets
|
||||||
|
swapped again when the CPT lands.
|
||||||
|
|
||||||
|
## Caveats
|
||||||
|
|
||||||
|
- Routing is one run per model, not three. The gaps between families are far larger
|
||||||
|
than the run-to-run spread seen on the talk fixture, but the 2B-vs-1.7B gap (62.3
|
||||||
|
vs 67.5) is not safe to call on one run.
|
||||||
|
- ~~The routing numbers only reach production once the LLM router is wired on. It is
|
||||||
|
still `nil`.~~ **Resolved the same evening:** the LLM router is wired at `voice.go:214`
|
||||||
|
behind `voice.llm_router`, the default is on, and `deploy/mavend.json` sets it `true`.
|
||||||
|
These numbers are the production path now, so the p50 ≈2.7s is a real per-turn cost and
|
||||||
|
not a bench artifact.
|
||||||
|
- ~~`/mnt/hdd1/llms/LFM2.5/Qwen3-1.7B-UD-Q4_K_XL.gguf` is a 293 MB truncated download
|
||||||
|
in the wrong directory.~~ **Deleted 2026-07-31.** The good 1.13 GB copy in `qwen3/` is
|
||||||
|
what `deploy/mavend.json` loads.
|
||||||
|
- Harness: `scratchpad/bakeoff.sh`, one server at a time, health-checked before each
|
||||||
|
run, `/v1/models` recorded per run. Never run two LLM consumers at once — see the
|
||||||
|
contamination note in `TALK-EVAL-31-07-2026.md`.
|
||||||
@@ -16,7 +16,7 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
|
|||||||
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
||||||
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
||||||
|
|
||||||
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test run-stt run-tts run-web download-embedder deps-go eval-router eval-recall
|
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||||
|
|
||||||
all: build
|
all: build
|
||||||
|
|
||||||
@@ -69,7 +69,20 @@ deps-go:
|
|||||||
done
|
done
|
||||||
$(GO) version
|
$(GO) version
|
||||||
|
|
||||||
test:
|
# fmt-check fails if any file needs gofmt. DESIGN.md has always said `make
|
||||||
|
# test` gates on gofmt and vet; it did not, so nine files quietly drifted.
|
||||||
|
# Run `gofmt -w` on whatever this prints.
|
||||||
|
fmt-check:
|
||||||
|
@bad=$$(gofmt -l internal cmd); \
|
||||||
|
if [ -n "$$bad" ]; then \
|
||||||
|
echo "these files need gofmt:"; echo "$$bad"; exit 1; \
|
||||||
|
fi
|
||||||
|
|
||||||
|
vet:
|
||||||
|
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||||
|
$(GO) vet ./internal/... ./cmd/...
|
||||||
|
|
||||||
|
test: fmt-check vet
|
||||||
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||||
$(GO) test -race -coverprofile=coverage.out ./internal/... ./cmd/...
|
$(GO) test -race -coverprofile=coverage.out ./internal/... ./cmd/...
|
||||||
|
|
||||||
@@ -90,6 +103,33 @@ eval-router:
|
|||||||
eval-recall:
|
eval-recall:
|
||||||
MAVEN_ONNX_LIB="$(MAVEN_ONNX_LIB)" $(GO) test -v -count=1 ./internal/memory/recalleval/
|
MAVEN_ONNX_LIB="$(MAVEN_ONNX_LIB)" $(GO) test -v -count=1 ./internal/memory/recalleval/
|
||||||
|
|
||||||
|
# eval-phrasing -- score nudge phrasing AND the conversational paths (chat,
|
||||||
|
# query, general knowledge) in internal/phraser/eval. Verbose so the
|
||||||
|
# report and every generated message land in the terminal. With no environment
|
||||||
|
# it scores the deterministic Stub only, which is what CI runs. Set
|
||||||
|
# MAVEN_LLM_URL to add the resident model:
|
||||||
|
# MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-phrasing
|
||||||
|
# The model run is slow (minutes) -- the timeout is raised to match. It covers
|
||||||
|
# two fixtures now (15 nudges + 27 conversational cases, and the chat replies are
|
||||||
|
# the long ones), hence 90m rather than 40m.
|
||||||
|
eval-phrasing:
|
||||||
|
$(GO) test -v -count=1 -timeout 90m ./internal/phraser/eval/
|
||||||
|
|
||||||
|
# eval-models — score ONE llama-server against the same fixture, for the
|
||||||
|
# resident-model bake-off (#278, #250). Start a server with the gguf you want,
|
||||||
|
# then:
|
||||||
|
#
|
||||||
|
# make eval-models MAVEN_LLM_URL=http://127.0.0.1:18100
|
||||||
|
#
|
||||||
|
# The report names carry the model llama-server reports, so runs from two
|
||||||
|
# checkpoints stay apart. Only the LLM test runs — the classifier baselines do
|
||||||
|
# not depend on the model and take the ONNX runtime with them.
|
||||||
|
MAVEN_LLM_URL ?= http://127.0.0.1:18099
|
||||||
|
|
||||||
|
eval-models:
|
||||||
|
MAVEN_LLM_URL="$(MAVEN_LLM_URL)" $(GO) test -v -count=1 -timeout 60m \
|
||||||
|
-run TestLLMRouterBaseline ./internal/router/eval/
|
||||||
|
|
||||||
run-stt: build-stt
|
run-stt: build-stt
|
||||||
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||||
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
||||||
@@ -115,9 +155,13 @@ deps-piper:
|
|||||||
-o /tmp/piper.tar.gz
|
-o /tmp/piper.tar.gz
|
||||||
tar -xzf /tmp/piper.tar.gz -C deps/
|
tar -xzf /tmp/piper.tar.gz -C deps/
|
||||||
|
|
||||||
EMBEDDER_DIR := $(shell pwd)/models/embedder
|
# multilingual-e5-small: an asymmetric retrieval model. It is trained to match
|
||||||
EMBEDDER_MODEL_URL := https://huggingface.co/Xenova/paraphrase-multilingual-MiniLM-L12-v2/resolve/main/onnx/model_quantized.onnx
|
# a short question against a longer passage, which is what note recall is.
|
||||||
EMBEDDER_TOKENIZER_URL := https://huggingface.co/Xenova/paraphrase-multilingual-MiniLM-L12-v2/resolve/main/tokenizer.json
|
# The quantized file is the one we download, deploy and measure — see
|
||||||
|
# RECALL-EVAL-31-07-2026.md.
|
||||||
|
EMBEDDER_DIR := $(shell pwd)/models/embedder/multilingual-e5-small
|
||||||
|
EMBEDDER_MODEL_URL := https://huggingface.co/Xenova/multilingual-e5-small/resolve/main/onnx/model_quantized.onnx
|
||||||
|
EMBEDDER_TOKENIZER_URL := https://huggingface.co/Xenova/multilingual-e5-small/resolve/main/tokenizer.json
|
||||||
|
|
||||||
download-embedder:
|
download-embedder:
|
||||||
mkdir -p $(EMBEDDER_DIR)
|
mkdir -p $(EMBEDDER_DIR)
|
||||||
|
|||||||
@@ -0,0 +1,169 @@
|
|||||||
|
# Phrasing evaluation — 31-07-2026
|
||||||
|
|
||||||
|
How Maven words a nudge, measured instead of argued. Counterpart to
|
||||||
|
`ROUTING-EVAL-31-07-2026.md`.
|
||||||
|
|
||||||
|
- Fixture + scorer: `internal/phraser/eval/` (`nudges_v1.json`, 15 cases; `eval.go`, `checks.go`)
|
||||||
|
- Reproduce: `MAVEN_LLM_URL=http://127.0.0.1:18099 make eval-phrasing`
|
||||||
|
- Model: Qwen3.5-0.8B Q4_K_M, the resident model. Not swapped.
|
||||||
|
- Commit: `a40bc55` (prompt fix)
|
||||||
|
|
||||||
|
Every check is a string or length test a human can read and disagree with. No model
|
||||||
|
grades another model here.
|
||||||
|
|
||||||
|
## Result
|
||||||
|
|
||||||
|
| | before | after |
|
||||||
|
|---|---|---|
|
||||||
|
| **cases passing every check** | **0/15** | **13/15** |
|
||||||
|
| mood in enum | 6/15 | 15/15 |
|
||||||
|
| Russian | 2/15 | 14/15 |
|
||||||
|
| length (≤120 chars, ≤16 words) | 13/15 | 15/15 |
|
||||||
|
| feminine self-reference | 15/15 | 15/15 |
|
||||||
|
| no cringe | 13/15 | 15/15 |
|
||||||
|
| on topic | 6/15 | 13/15 |
|
||||||
|
| p50 latency | 11.4s | 11.4s |
|
||||||
|
|
||||||
|
Latency did not move and is not good. 11s to word one nudge on this box.
|
||||||
|
|
||||||
|
## The bug reproduced
|
||||||
|
|
||||||
|
Yes, exactly as reported. 7 of 15 messages were the literal string `"..."`, and one was
|
||||||
|
`"full voice message"`. Both are text copied straight out of the prompt.
|
||||||
|
|
||||||
|
The system prompt said:
|
||||||
|
|
||||||
|
```
|
||||||
|
Respond ONLY with valid JSON: {"response": "full voice message", "mood": "neutral"}
|
||||||
|
```
|
||||||
|
|
||||||
|
and the user prompt said:
|
||||||
|
|
||||||
|
```
|
||||||
|
Respond as JSON: {"response": "...", "mood": "..."}
|
||||||
|
```
|
||||||
|
|
||||||
|
A 0.8B does not read `"..."` as "put your answer here". It reads it as the answer. The
|
||||||
|
prompt was a worked example whose worked part was blank, so the model filled the slot by
|
||||||
|
copying. This is the whole of finding 1.
|
||||||
|
|
||||||
|
## What else was wrong
|
||||||
|
|
||||||
|
Four separate faults, all prompt-side:
|
||||||
|
|
||||||
|
1. **Placeholder echo** (7 cases) — above.
|
||||||
|
2. **Wrong language** (13/15 failed the language check). The prompt was entirely English
|
||||||
|
and said "in the user's language (Russian or English)". The model picked English. It is
|
||||||
|
never English: the nudge is spoken by a Russian piper voice.
|
||||||
|
3. **Rule names are English identifiers.** `netdata_critical`, `service_down`, `break` went
|
||||||
|
into the prompt raw. The model cannot nudge about a topic it has not been told in words,
|
||||||
|
so 9/15 were off topic. The daemon knows what its own rules mean; now it says so.
|
||||||
|
4. **Mood invented** (`"warm"`, twice). The enum was listed in a parenthesis at the end of
|
||||||
|
an English sentence. Now it is its own line: "ровно одно из: neutral, happy, thinking,
|
||||||
|
tired, confused."
|
||||||
|
|
||||||
|
Plus two non-prompt faults the run exposed:
|
||||||
|
|
||||||
|
- **The no-parse fallback was English.** When the model returned nothing usable, the body
|
||||||
|
became `fmt.Sprintf("%s — %s", rule, sev)` — `"water — care"` — and that string went to
|
||||||
|
a Russian TTS. Now it falls back to plain Russian.
|
||||||
|
- **Durations were English.** `humanDur` returns "3 hours"; it was landing verbatim inside
|
||||||
|
Russian sentences. Nudges now use a Russian formatter.
|
||||||
|
|
||||||
|
## Three iterations, and what each taught
|
||||||
|
|
||||||
|
| | score | change |
|
||||||
|
|---|---|---|
|
||||||
|
| baseline | 0/15 | — |
|
||||||
|
| iter 1 | 2/15 | Russian prompt, filled-in examples, Russian durations |
|
||||||
|
| iter 2 | 11/15 | required keyword per rule, one example instead of five, Russian fallback |
|
||||||
|
| iter 3 | **13/15** | examples moved to topics that are not rules |
|
||||||
|
|
||||||
|
The interesting step is 1 → 2. Fixing the placeholder did not fix the disease, it moved it:
|
||||||
|
the model stopped copying `"..."` and started copying my first example instead. Five nudges
|
||||||
|
in a row came back as `"Ты не пил воду три часа. Налей стакан."` regardless of the rule.
|
||||||
|
|
||||||
|
**A small model copies the nearest concrete text in its prompt.** That is one failure mode
|
||||||
|
with two symptoms. The fix that stuck was making the examples about laundry and a laptop
|
||||||
|
battery — topics no rule ever produces, so copying them is visible in the score rather than
|
||||||
|
invisibly passing the water cases.
|
||||||
|
|
||||||
|
## Do not oversell 13/15
|
||||||
|
|
||||||
|
Seven of the thirteen passes are the **deterministic fallback**, not the model:
|
||||||
|
`"Напоминаю: таблетки."`, `"Сервис не отвечает."`, `"Критический алярм: проверь диск."`,
|
||||||
|
`"Ты давно не пил воду."`. Those are strings this commit added to Go. The model returned
|
||||||
|
nothing parseable and the fallback scored.
|
||||||
|
|
||||||
|
So the honest reading is roughly **6/15 from the model, 7/15 from a fallback, 2/15 failing**.
|
||||||
|
The prompt fix is real — `"..."` is nearly gone and the language and mood checks are clean —
|
||||||
|
but a large part of the jump is that failure now degrades into Russian instead of into
|
||||||
|
`"water — care"`. That is a genuine improvement for the operator and a weak one for the model.
|
||||||
|
|
||||||
|
The two remaining failures: one `"..."` recurrence (`routine-stretch`) and one meal nudge
|
||||||
|
that never says food.
|
||||||
|
|
||||||
|
## Tried and reverted: an example-led nudge prompt (#393)
|
||||||
|
|
||||||
|
The idea was that a 0.8B copies examples better than it follows rules, so the nudge prompt
|
||||||
|
was rewritten to lead with five on-topic examples (water, break, pills, morning, service) and
|
||||||
|
the prose rules were compressed to pay for the tokens: 1190 chars down to 986.
|
||||||
|
|
||||||
|
It measured **worse**, three runs each side, same llama-server, same fixture:
|
||||||
|
|
||||||
|
| run | before | after |
|
||||||
|
|---|---|---|
|
||||||
|
| 1 | 12/15 (address 14) | 11/15 (address 13) |
|
||||||
|
| 2 | 13/15 (address 15) | 12/15 (address 15) |
|
||||||
|
| 3 | 14/15 (address 15) | 11/15 (address 12) |
|
||||||
|
|
||||||
|
`feminine` and `hisgender` were 15/15 on all six runs, so they measure nothing here. The
|
||||||
|
regression is all in `address`: 44/45 before, 40/45 after. Formal "вы"/"ваше" and plural
|
||||||
|
imperatives came back, and so did `"..."`.
|
||||||
|
|
||||||
|
Two likely causes, both about the same thing — **examples do not carry a prohibition**. The
|
||||||
|
old prompt spent a whole sentence on «говоришь на "ты", в единственном числе»; the new one
|
||||||
|
demoted that to one item in a long "никогда" list, and the model stopped obeying it. And
|
||||||
|
making the examples on-topic let their *wording* leak: a break case came back as
|
||||||
|
«Вы давно не пили воду. Выпей стакан.» — the water example, verbatim, in the wrong slot.
|
||||||
|
That is exactly the failure the laundry/laptop examples were chosen to avoid.
|
||||||
|
|
||||||
|
Change reverted. What survives is the measurement: a rule the model must obey needs its own
|
||||||
|
sentence, and examples must stay off-topic. Also note the before side alone spans 12–14 of
|
||||||
|
15 — this fixture cannot resolve anything smaller than about three cases.
|
||||||
|
|
||||||
|
## Broken, found, not fixed
|
||||||
|
|
||||||
|
1. ~~**`checkFeminine` only catches half the constraint.**~~ **Fixed** (#381). It scanned for
|
||||||
|
masculine self-reference only, so three messages that addressed the *owner* in the feminine
|
||||||
|
("ты давно не отдыхал**а**") scored clean. There is now a second check, `hisgender`: a
|
||||||
|
feminine past-tense verb (-ла/-лась) in a sentence addressed to him ("ты", "тебе", "твой")
|
||||||
|
fails, unless the verb is hers ("я заметила", "напомнила тебе"). It is a suffix rule, not a
|
||||||
|
parser — see the comment in `checks.go` for what it misses. A fresh 15-case run after adding
|
||||||
|
it scored **12/15** with `hisgender` 15/15; the model did not repeat the feminine address in
|
||||||
|
that sample, and the check is pinned by unit tests on the recorded bad strings instead.
|
||||||
|
2. **Grammar is not checked at all, and it is bad.** `"Он не ел 11 дней"` (it was 11 hours),
|
||||||
|
`"Сонуждились 7 дней"` (not a word), `"Они забыли воду"` (wrong person entirely). Every
|
||||||
|
one of these passes all six checks. The fixture measures properties, not fluency, and at
|
||||||
|
0.8B fluency is the binding constraint.
|
||||||
|
3. **Unit confusion.** The model turns hours into days about a third of the time. The
|
||||||
|
prompt now says "11 ч"; it reads it as days.
|
||||||
|
4. **11s p50.** Unchanged and untouched here. A nudge the model takes eleven seconds to
|
||||||
|
word has missed its moment. Worth its own task.
|
||||||
|
5. **The keyword hint is close to teaching to the test.** `ruleKeywords` names the word the
|
||||||
|
on-topic check looks for. It is defensible — the daemon genuinely knows its rule topics
|
||||||
|
and the model genuinely cannot infer them from `netdata_critical` — but the on-topic
|
||||||
|
number is softer than the others because of it.
|
||||||
|
|
||||||
|
## Next steps
|
||||||
|
|
||||||
|
1. ~~**Add a second-person gender check**~~ — done, `hisgender` in `checks.go` (#381).
|
||||||
|
2. **Decide whether the fallback should count as a pass.** Right now `Score` cannot tell a
|
||||||
|
model answer from a fallback. Either mark fallback bodies in `PhrasedNudge` or count them
|
||||||
|
in their own column. Without that, any future prompt change can score well by failing
|
||||||
|
more.
|
||||||
|
3. **Attack the 11s.** Nudge phrasing is short and non-interactive; thinking off is the first
|
||||||
|
thing to try, as it was for routing (#376).
|
||||||
|
4. **Re-measure when #122 lands.** The CPT'd Qwen3-1.7B is the target. 13/15 with seven
|
||||||
|
fallbacks is the floor it has to beat, and the fluency problems above are the ones a
|
||||||
|
bigger, Russian-trained checkpoint should actually fix.
|
||||||
@@ -90,4 +90,7 @@ later* is the worker + RAG.
|
|||||||
4. **Deferred work** — larger reasoner, custom Piper voice and other expansions.
|
4. **Deferred work** — larger reasoner, custom Piper voice and other expansions.
|
||||||
|
|
||||||
## Non-goals (unchanged)
|
## Non-goals (unchanged)
|
||||||
Never phones home. Not a nag. Not autonomous. Feminine-gendered RU self-ref.
|
Not a nag. Not autonomous. Feminine-gendered RU self-ref. No telemetry, no
|
||||||
|
cloud model, no third-party account — but she MAY read external sources to
|
||||||
|
answer world questions (Kiwix first, search optional). "Never phones home" as
|
||||||
|
an absolute is deprecated, owner's call 2026-07-31; see CLAUDE.md § Non-goals.
|
||||||
|
|||||||
+153
-2
@@ -75,6 +75,14 @@ same vector. A note is indexed in both places with the same embedding, so if it
|
|||||||
`QueryNotes` it fails again here — the branch can only ever return a **fact**. Its comment calls it
|
`QueryNotes` it fails again here — the branch can only ever return a **fact**. Its comment calls it
|
||||||
"additive"; for notes it is not.
|
"additive"; for notes it is not.
|
||||||
|
|
||||||
|
**Fixed (Vikunja #373).** The memory pass now runs *first*, as one search over notes and facts with
|
||||||
|
one gate, so whichever memory is clearly the best match answers — note or fact. The notes-only pass
|
||||||
|
stays behind it for notes the vector index does not hold. No threshold changed, so the set of
|
||||||
|
questions Maven answers is the same; only which memory answers them. The fixture gained two mixed
|
||||||
|
note+fact cases (`ru-mixed-031`, `ru-mixed-032`), which is why the counts below are out of 27
|
||||||
|
answerable cases and not 25: hash recall@1 36.0% (9/25) → 37.0% (10/27), e5 recall@1 72.0% (18/25) →
|
||||||
|
70.4% (19/27) with answered-after-gate 68.0% → 66.7% and false recall unchanged at 1/5.
|
||||||
|
|
||||||
### 5. Ranking has no recency or type signal, and the store is not the bottleneck
|
### 5. Ranking has no recency or type signal, and the store is not the bottleneck
|
||||||
|
|
||||||
`internal/store/notes.go:67` sorts by cosine and uses `ts` only to break an exact float tie, which
|
`internal/store/notes.go:67` sorts by cosine and uses `ts` only to break an exact float tie, which
|
||||||
@@ -83,14 +91,157 @@ sqlite-backed `store.MemoryStore` and `memory.InMemoryStore` identically — bot
|
|||||||
(`internal/store/memory.go:64`) at ~150µs over 42 rows against a ~59ms query embed. An ANN index is
|
(`internal/store/memory.go:64`) at ~150µs over 42 rows against a ~59ms query embed. An ANN index is
|
||||||
not the problem to solve.
|
not the problem to solve.
|
||||||
|
|
||||||
|
## Re-measured after the embedder swap — 31-07-2026, later the same day
|
||||||
|
|
||||||
|
Changed: `models/embedder/` is now **multilingual-e5-small** (quantized, 118MB), with `query: ` in
|
||||||
|
front of a question and `passage: ` in front of a stored note (Vikunja #371). `deploy/mavend.json`
|
||||||
|
and `make download-embedder` now name the same file, and it is the quantized one — that is what the
|
||||||
|
column below measures (Vikunja #372). Everything else is unchanged: same fixture, same store, same
|
||||||
|
0.55 gate. The old column is the baseline and is left as it was.
|
||||||
|
|
||||||
|
| | recall+onnx, MiniLM (baseline) | recall+onnx, e5-small (new) |
|
||||||
|
|---|---|---|
|
||||||
|
| **recall@1** | 60.0% (15/25) | **72.0% (18/25)** |
|
||||||
|
| recall@3 | 80.0% (20/25) | 84.0% (21/25) |
|
||||||
|
| **answered after the 0.55 gate** | 48.0% (12/25) | **72.0% (18/25)** |
|
||||||
|
| wrong note on top / tie on top | 10 / 0 | 7 / 0 |
|
||||||
|
| ranked first, then silenced by the gate | 3 | 0 |
|
||||||
|
| **false recall** | 1/5 (20%) | **5/5 (100%)** |
|
||||||
|
| top-1 score when right, min / median | 0.559 / 0.678 | 0.791 / 0.857 |
|
||||||
|
| top-1 when it must stay silent, median / max | 0.470 / 0.567 | 0.815 / 0.835 |
|
||||||
|
| RU / EN / `hard` cases passed | 13/24 / 3/6 / 2/11 | 14/24 / 4/6 / 5/11 |
|
||||||
|
| latency p50 / p95 / max | 59ms / 148ms / 194ms | 18ms / 37ms / 49ms |
|
||||||
|
|
||||||
|
### What moved
|
||||||
|
|
||||||
|
Ranking got better and got faster. Half the previously-unwinnable `hard` cases now pass (2/11 →
|
||||||
|
5/11), the guitar note no longer beats the docker-logs note, and the gate stops silencing notes that
|
||||||
|
already ranked first. The quantized e5 is also ~3x quicker than the fp32 MiniLM it replaces.
|
||||||
|
|
||||||
|
### What got worse: the gate is now a no-op
|
||||||
|
|
||||||
|
e5 packs every cosine into a narrow high band. Right-note scores start at 0.791; must-stay-silent
|
||||||
|
scores reach 0.835. **The distributions still overlap, and now they overlap above the gate**, so
|
||||||
|
0.55 admits everything and false recall goes from 1/5 to 5/5. The sweep:
|
||||||
|
|
||||||
|
```
|
||||||
|
gate 0.50–0.70: answered 18/25 (72%) false recall 5/5
|
||||||
|
gate 0.80: answered 17/25 (68%) false recall 4/5
|
||||||
|
gate 0.90: answered 0/25 ( 0%) false recall 0/5
|
||||||
|
```
|
||||||
|
|
||||||
|
There is no value that keeps real recall and rejects made-up questions — same conclusion as before,
|
||||||
|
now with a wider band and no room at all. `query_min_score` was left at 0.55 as instructed. **The
|
||||||
|
recommendation is to leave it there and stop tuning it**: any number under ~0.79 is a no-op and
|
||||||
|
anything above starts cutting real recall long before it stops the false ones. The fix is a margin
|
||||||
|
gate (`top1 − top2 > δ`), next-steps item 3, which is now the top item.
|
||||||
|
|
||||||
|
### The prefixes did not do the work
|
||||||
|
|
||||||
|
A control run with both prefixes set to the empty string scored the **same** recall@1 (72%), a
|
||||||
|
slightly better recall@3 (88%) and the same 5/5 false recall. So on this fixture the gain comes from
|
||||||
|
the model, not from the `query:` / `passage:` split. The prefixes are kept because they are how e5
|
||||||
|
was trained and the split is the right shape for the read path, but they are not worth defending on
|
||||||
|
this evidence — a bigger fixture may say otherwise.
|
||||||
|
|
||||||
|
### Stored vectors from the old model are now junk
|
||||||
|
|
||||||
|
Cosine between a MiniLM vector and an e5 vector means nothing. Every row already in `notes` and in
|
||||||
|
the vector memory table was written by the old model, so after this deploy they will score as noise
|
||||||
|
against a new query. A live database needs every note and fact re-embedded before recall works at
|
||||||
|
all. Filed as its own task.
|
||||||
|
|
||||||
|
## Margin gate — 31-07-2026, third run
|
||||||
|
|
||||||
|
Next-steps item 3, done. The absolute gate is replaced by a **margin gate**: answer only when the
|
||||||
|
top hit beats the runner-up by more than delta (`top1 − top2 > δ`). Same fixture, same e5 embedder,
|
||||||
|
same store as the run above. `internal/memory/gate.go` holds the check; both read paths call it
|
||||||
|
(`cmd/mavend/recall.go` and the notes-RAG branch in `voice.go`). New knob `voice.query_min_margin`
|
||||||
|
in `deploy/mavend.json`, default 0.008.
|
||||||
|
|
||||||
|
### Why the absolute gate could not work, in one line of data
|
||||||
|
|
||||||
|
The harness now prints the margin distributions, and they barely overlap where the raw scores
|
||||||
|
overlap completely:
|
||||||
|
|
||||||
|
| | top-1 score | margin (top1 − top2) |
|
||||||
|
|---|---|---|
|
||||||
|
| right note first (n=18) | min 0.810, median 0.862, max 0.890 | min 0.001, median 0.029, max 0.053 |
|
||||||
|
| must stay silent (n=5) | min 0.795, median 0.815, max 0.835 | min 0.000, median 0.002, **max 0.019** |
|
||||||
|
|
||||||
|
Four of the five must-be-silent cases have a margin at or under 0.002 — when there is nothing to
|
||||||
|
recall, e5 finds several notes equally close and no clear winner. That is the signal the absolute
|
||||||
|
score throws away.
|
||||||
|
|
||||||
|
### The delta sweep
|
||||||
|
|
||||||
|
Absolute gate held at 0.55 throughout.
|
||||||
|
|
||||||
|
```
|
||||||
|
delta 0.000: answered 18/25 (72%) false recall 5/5
|
||||||
|
delta 0.002: answered 17/25 (68%) false recall 3/5
|
||||||
|
delta 0.005: answered 17/25 (68%) false recall 2/5
|
||||||
|
delta 0.008: answered 17/25 (68%) false recall 1/5 <- chosen
|
||||||
|
delta 0.010: answered 15/25 (60%) false recall 1/5
|
||||||
|
delta 0.012: answered 14/25 (56%) false recall 1/5
|
||||||
|
delta 0.015: answered 12/25 (48%) false recall 1/5
|
||||||
|
delta 0.020: answered 11/25 (44%) false recall 0/5
|
||||||
|
delta 0.025: answered 9/25 (36%) false recall 0/5
|
||||||
|
delta 0.030: answered 8/25 (32%) false recall 0/5
|
||||||
|
delta 0.040: answered 4/25 (16%) false recall 0/5
|
||||||
|
delta 0.050: answered 2/25 ( 8%) false recall 0/5
|
||||||
|
delta 0.060: answered 0/25 ( 0%) false recall 0/5
|
||||||
|
```
|
||||||
|
|
||||||
|
### Chosen: δ = 0.008
|
||||||
|
|
||||||
|
It is the best point on the frontier, not a taste call. **0.008 dominates 0.010, 0.012 and 0.015
|
||||||
|
outright** — same 1/5 false recall, 8 to 20 points more real recall. Everything below it buys recall
|
||||||
|
back only by admitting more false recalls (0.005 → 2/5, 0.002 → 3/5). The next real improvement is
|
||||||
|
0.020 at 0/5 false, and it costs 24 points of recall to get there.
|
||||||
|
|
||||||
|
The brief's bar was "recall above 60% with false recall at 1/5 or better". 0.008 clears it with room:
|
||||||
|
68% and 1/5.
|
||||||
|
|
||||||
|
### Before / after
|
||||||
|
|
||||||
|
| | absolute gate 0.55 (previous) | margin gate δ=0.008 |
|
||||||
|
|---|---|---|
|
||||||
|
| recall@1 (ranking, ungated) | 72.0% (18/25) | 72.0% (18/25) — unchanged, the gate does not rank |
|
||||||
|
| **answered after the gate** | 72.0% (18/25) | **68.0% (17/25)** |
|
||||||
|
| **false recall** | **5/5 (100%)** | **1/5 (20%)** |
|
||||||
|
| fixture cases passed | 18/30 | **21/30** |
|
||||||
|
|
||||||
|
Four false recalls removed for one real answer. That is the trade the spec asks for — she is not a
|
||||||
|
guesser-of-truth. The one survivor is `en-pref-025` ("should i be offered wine"), which recalls a
|
||||||
|
filler note at 0.796 with a 0.019 margin: the widest silent-case margin in the fixture, and it sits
|
||||||
|
inside the real-recall range, so no delta removes it without taking real answers with it.
|
||||||
|
|
||||||
|
### Does the absolute cutoff still earn its keep? Marginally — kept
|
||||||
|
|
||||||
|
On this fixture with e5 it is a **no-op**: the lowest right-note score is 0.791, so 0.55 rejects
|
||||||
|
nothing the margin does not already reject. It is kept for two reasons, neither glamorous. It still
|
||||||
|
does real work for the hash embedder (its own sweep shows answers dropping from 16% to 0% between
|
||||||
|
0.30 and 0.50), and it is the only thing standing between the user and a reply built from a store
|
||||||
|
where everything is far away but one row happens to be a little less far — a near-empty database, or
|
||||||
|
the stale-vector case below. Cheap insurance, no measured cost. If a later embedder makes it bite,
|
||||||
|
the sweep is one command.
|
||||||
|
|
||||||
|
### Caveat on the numbers
|
||||||
|
|
||||||
|
Five must-be-silent cases is a thin basis for a 4-point decision. 1/5 and 2/5 differ by one case.
|
||||||
|
The shape of the frontier is trustworthy — margins separate, absolute scores do not — but δ=0.008
|
||||||
|
itself should be re-read off a bigger fixture (next-steps item 6) before anyone defends the third
|
||||||
|
decimal.
|
||||||
|
|
||||||
## Next steps — ordered by value-to-risk; nothing here is a decision
|
## Next steps — ordered by value-to-risk; nothing here is a decision
|
||||||
|
|
||||||
1. **Swap the embedder to `multilingual-e5-small` with `query:`/`passage:` prefixes.** One config
|
1. **Swap the embedder to `multilingual-e5-small` with `query:`/`passage:` prefixes.** One config
|
||||||
change plus a prefix in `onnxembedder.go`, re-measurable in one command.
|
change plus a prefix in `onnxembedder.go`, re-measurable in one command.
|
||||||
2. **Re-run `make eval-recall`, then set the gate from the sweep** — not before. Any
|
2. **Re-run `make eval-recall`, then set the gate from the sweep** — not before. Any
|
||||||
`query_min_score` picked against today's embedder describes a model on its way out.
|
`query_min_score` picked against today's embedder describes a model on its way out.
|
||||||
3. **Replace the absolute-score gate with a margin gate** (`top1 − top2 > δ`) — as the routing eval
|
3. ~~**Replace the absolute-score gate with a margin gate**~~ — done, see the section above.
|
||||||
concluded, absolute cosine cannot see a flat distribution.
|
δ=0.008, false recall 5/5 → 1/5.
|
||||||
4. **Delete or repair the dead `memStore` branch** at `voice.go:776` — search before the gate,
|
4. **Delete or repair the dead `memStore` branch** at `voice.go:776` — search before the gate,
|
||||||
gate it separately, or restrict it to facts and say so.
|
gate it separately, or restrict it to facts and say so.
|
||||||
5. **Add a mild time decay to ranking** — the newest statement of a preference is the true one.
|
5. **Add a mild time decay to ranking** — the newest statement of a preference is the true one.
|
||||||
|
|||||||
+138
-5
@@ -40,6 +40,139 @@ model → classifier as failure floor.
|
|||||||
|
|
||||||
Never compare a hash-embedder run to an ONNX one.
|
Never compare a hash-embedder run to an ONNX one.
|
||||||
|
|
||||||
|
## Re-measured after the prompt fix
|
||||||
|
|
||||||
|
The table above is the **baseline at commit `46259b4`**, kept as-is. The prompt fix (query
|
||||||
|
tested before fact, plus `repeat_penalty` and a bounded grammar string) was then measured on
|
||||||
|
an otherwise idle box — no other eval sharing llama-server, so these latencies are real
|
||||||
|
rather than contention.
|
||||||
|
|
||||||
|
| | llm-only (0.8B) | cascade+llm (0.8B) | llm-only, thinking off |
|
||||||
|
|---|---|---|---|
|
||||||
|
| **intent-only accuracy** | 48.7% → **61.8%** | 50.0% → **63.2%** | **67.1%** |
|
||||||
|
| full accuracy (intent+slots+gate) | 23.7% → **38.2%** | 32.9% → **47.4%** | **42.1%** |
|
||||||
|
| route errors | 2 → **0** | 0 → 0 | **0** |
|
||||||
|
| p50 / p95 latency | **1.08s / 1.55s** | **1.04s / 1.53s** | **0.93s / 1.41s** |
|
||||||
|
|
||||||
|
Three things this run settles:
|
||||||
|
|
||||||
|
1. **The prompt fix holds.** An earlier contended run reported 60.5% / 36.8% for llm-only;
|
||||||
|
the quiet run gives 61.8% / 38.2%. Close enough to call the gain real, and the earlier
|
||||||
|
run's 4-5s latency figures were contention, not the model.
|
||||||
|
2. **`query→fact` fell from ×15 to ×7**, and both unparseable replies are gone. Zero route
|
||||||
|
errors in every LLM configuration.
|
||||||
|
3. **`note→fact ×4` is real, not noise.** It shows up in the quiet run too. The agent that
|
||||||
|
wrote the prompt fix suspected its own change might have caused it by pulling assertive
|
||||||
|
`запиши что…` phrasings toward fact, and that suspicion stands — all five `ru-note-*`
|
||||||
|
cases now land on fact. Tracked as Vikunja #375.
|
||||||
|
|
||||||
|
The `thinking off` column above read as the best configuration measured so far (Vikunja #376).
|
||||||
|
**It was wrong** — see the controlled re-run below. Ignore that column.
|
||||||
|
|
||||||
|
Still `6 / 6` missed clarify — the router has no way to say "I don't know" (Vikunja #359).
|
||||||
|
That is unchanged by anything here.
|
||||||
|
|
||||||
|
## Thinking off — 31-07-2026, controlled re-run (Vikunja #376)
|
||||||
|
|
||||||
|
The "thinking off wins by 6 points" observation above **does not hold**. It was a measurement
|
||||||
|
artefact, and the earlier table's `thinking off` column should be ignored.
|
||||||
|
|
||||||
|
The thinking-off variant was scored by a hand-rolled HTTP client living in the test file
|
||||||
|
instead of `llm.Client`. That copy did not send `repeat_penalty`, which the real router does
|
||||||
|
send (`routeRepeatPenalty = 1.15`). So the two columns differed on two axes at once, and the
|
||||||
|
one that mattered was the penalty, not the thinking mode.
|
||||||
|
|
||||||
|
Re-measured with everything else held equal — same fixture, same prompt, same grammar, same
|
||||||
|
sampling, same idle box, the three configurations run back to back and never concurrently:
|
||||||
|
|
||||||
|
| | llm-only, thinking on | llm-only, thinking off | cascade+llm |
|
||||||
|
|---|---|---|---|
|
||||||
|
| intent-only accuracy | 59.2% (45/76) | 59.2% (45/76) | 61.8% (47/76) |
|
||||||
|
| full accuracy (intent+slots+gate) | 38.2% (29/76) | 38.2% (29/76) | 57.9% (44/76) |
|
||||||
|
| route errors | 3 | 3 | 0 |
|
||||||
|
| grammar violations | 3 (all 3 route errors) | 3 (same 3 cases) | 0 |
|
||||||
|
| missed clarify | 5 / 6 | 5 / 6 | 5 / 6 |
|
||||||
|
| p50 latency | 836ms | 920ms | 810ms |
|
||||||
|
| p95 latency | 1.41s | 2.00s | 1.31s |
|
||||||
|
|
||||||
|
Thinking off is not just a tie on the headline numbers — it is identical case for case, with
|
||||||
|
the same confusion matrix and the same three unparseable replies. The latency difference is
|
||||||
|
run-to-run noise on one box, and it points the wrong way here.
|
||||||
|
|
||||||
|
The reason is simpler than any accuracy argument: **this llama-server build ignores the
|
||||||
|
request-level thinking switch for this model.** Probed directly against the running server
|
||||||
|
with `chat_template_kwargs.enable_thinking = false`, `chat_template_kwargs.thinking = false`
|
||||||
|
and top-level `reasoning_budget = 0` — all three return a byte-identical answer with the
|
||||||
|
thinking trace still in `reasoning_content`, and the server reports the prompt prefix as
|
||||||
|
cached, meaning the rendered template did not change. There was never anything being turned
|
||||||
|
off, which is also why the numbers match exactly.
|
||||||
|
|
||||||
|
Nothing was defaulted. `internal/llm` still has no `chat_template_kwargs` field, `VoiceConfig`
|
||||||
|
has no thinking flag, and `deploy/mavend.json` is unchanged. The misleading third
|
||||||
|
configuration is removed from `internal/router/eval` so the table it produced cannot be quoted
|
||||||
|
again.
|
||||||
|
|
||||||
|
Two caveats worth saying out loud:
|
||||||
|
|
||||||
|
- **The fixture is 76 cases.** A 6-point difference on 76 cases is roughly 4-5 cases and would
|
||||||
|
not have been worth trusting even if it had reproduced. This one was exactly 0 cases, which
|
||||||
|
is a much easier call.
|
||||||
|
- **This is one server build and one checkpoint** (`b9351`, Qwen3.5-0.8B Q4_K_M). If the
|
||||||
|
#122 checkpoint or a newer llama.cpp does honour the switch, the question reopens — but it
|
||||||
|
reopens as an unmeasured question, not as a 6-point win.
|
||||||
|
|
||||||
|
Phrasing was **not** measured. Whether thinking helps there is still open, and now also blocked
|
||||||
|
on the same "can we even turn it off" question.
|
||||||
|
|
||||||
|
## Clock and calendar rule — 31-07-2026 (Vikunja #374)
|
||||||
|
|
||||||
|
`routeSystem` never said whether "который час" or "какое число завтра" are `system` or
|
||||||
|
`query`, and `system→query ×4` showed up in every run. The rule added says: the clock and the
|
||||||
|
calendar date themselves are `system`; what is *written in* the calendar or in memory
|
||||||
|
("что у меня завтра", "какие есть напоминания") stays `query`; and a time named inside a
|
||||||
|
request ("напомни завтра…") is just a detail of the request, not a reason for `system`.
|
||||||
|
|
||||||
|
That split is not a preference. In `cmd/mavend/voice.go` only `replySystem` owns the clock and
|
||||||
|
the date formatter, so a clock question routed to `query` falls into the embedder + note RAG
|
||||||
|
and answers "не знаю". The agenda, on the other hand, is answered by `ParseCalendarDate` +
|
||||||
|
`CalendarEvents` *inside* the `query` branch, so that side has to stay `query`. The rule sits
|
||||||
|
above the question test because every one of these utterances carries a question word and a
|
||||||
|
later rule would never be reached.
|
||||||
|
|
||||||
|
The fixture is now 77 cases: one calendar-agenda case was added
|
||||||
|
(`ru-query-019` "что у меня стоит в календаре на послезавтра", intent `query`) specifically so
|
||||||
|
an over-broad system rule cannot pass unnoticed. The clock/date cases (`ru-sys-001/002/005`,
|
||||||
|
`en-sys-001`) already existed.
|
||||||
|
|
||||||
|
Three runs, same box, back to back, never concurrently:
|
||||||
|
|
||||||
|
| | baseline | first rule (too broad) | rule as committed |
|
||||||
|
|---|---|---|---|
|
||||||
|
| llm-only intent-only | 59.2% (45/76) | 54.5% (42/77) | 59.7% (46/77) |
|
||||||
|
| llm-only full | 38.2% | 35.1% | 39.0% |
|
||||||
|
| llm-only route errors | 3 | 4 | 5 |
|
||||||
|
| llm-only p50 | 1.09s | 0.91s | 0.93s |
|
||||||
|
| cascade+llm intent-only | 61.8% (47/76) | 58.4% | 62.3% (48/77) |
|
||||||
|
| cascade+llm full | 57.9% | 54.5% | 59.7% |
|
||||||
|
| cascade+llm route errors | 0 | 0 | 0 |
|
||||||
|
| cascade+llm p50 | 0.91s | 0.80s | 1.04s |
|
||||||
|
|
||||||
|
**The targeted bug is fixed and the headline number did not move.** `system→query ×4` is gone
|
||||||
|
in both LLM configurations — the `time` and `date` tags go from 0/2 and 0/2 to 2/2 and 2/2 —
|
||||||
|
but the model then over-applies the rule, and `query→system ×5` plus `reminder→system ×2`
|
||||||
|
appear where they did not exist before. Net accuracy is a wash, inside the noise of a 77-case
|
||||||
|
fixture.
|
||||||
|
|
||||||
|
The first attempt is shown because it is the honest history: it said "спрашивает время, дату
|
||||||
|
или день недели → system" with no scope, which swept up reminders, and it cost 3-5 points. It
|
||||||
|
was tightened once, on the reasoning that a rule capturing "напомни завтра в 7" is simply
|
||||||
|
wrong, and not tuned further. The remaining `query/reminder → system` over-trigger is a new,
|
||||||
|
separate weakness of the sub-1B model and deserves its own task rather than more prompt
|
||||||
|
kneading against a held-out fixture.
|
||||||
|
|
||||||
|
The rule is kept. It is correct about what the daemon can answer, and the failure it replaces
|
||||||
|
was silent ("не знаю" to "который час") while the one it introduces is loud.
|
||||||
|
|
||||||
## Findings
|
## Findings
|
||||||
|
|
||||||
### 1. The resident model does route better — 50.0% vs 36.8%
|
### 1. The resident model does route better — 50.0% vs 36.8%
|
||||||
@@ -110,11 +243,11 @@ Note the grammar's `string ::= "\"" ([^"\\] | "\\" .)* "\""` is unbounded, so no
|
|||||||
|
|
||||||
### 7. Two hypotheses tested and closed
|
### 7. Two hypotheses tested and closed
|
||||||
|
|
||||||
- **Thinking mode is a non-issue.** Qwen3.5's template defaults `thinking = 1`, so
|
- **Thinking mode is a non-issue.** Confirmed twice now, the second time properly — see the
|
||||||
grammar-constrained JSON lands in `reasoning_content` with `content` empty —
|
controlled re-run section. Grammar-constrained JSON lands in `reasoning_content` with
|
||||||
`llm.Client`'s fallback handles it. A `thinking off` run scored *identically* (18/76,
|
`content` empty and `llm.Client`'s fallback handles it; the request-level switch does
|
||||||
48.7%, same p50). `internal/llm` deliberately does **not** grow a `chat_template_kwargs`
|
nothing on this build. `internal/llm` deliberately does **not** grow a
|
||||||
field.
|
`chat_template_kwargs` field.
|
||||||
- **Runaway array repetition does not reproduce.** An isolated smoke test with a stripped
|
- **Runaway array repetition does not reproduce.** An isolated smoke test with a stripped
|
||||||
grammar emitted `{"intent":"reminder"}` until `MaxTokens`; under the real `routeSystem`
|
grammar emitted `{"intent":"reminder"}` until `MaxTokens`; under the real `routeSystem`
|
||||||
prompt the few-shot examples anchor it to one object. 2 errors in 76, not 76.
|
prompt the few-shot examples anchor it to one object. 2 errors in 76, not 76.
|
||||||
|
|||||||
@@ -0,0 +1,150 @@
|
|||||||
|
# Conversational phrasing eval — 31-07-2026
|
||||||
|
|
||||||
|
Every score measured tonight, on the three paths the nudge eval never touched:
|
||||||
|
chat, query-with-notes, and general knowledge.
|
||||||
|
|
||||||
|
**Short version: the plumbing got fixed and the score barely moved.** Grammar and
|
||||||
|
Russian prompts together took the composite from ~9 to ~14 of 27. Everything
|
||||||
|
still failing is the model not knowing things or not holding a constraint, and
|
||||||
|
prompting is out of levers. Settles the measurement half of Vikunja #395 / #398 /
|
||||||
|
#400.
|
||||||
|
|
||||||
|
## How to reproduce
|
||||||
|
|
||||||
|
```sh
|
||||||
|
# llama-server: -c 4096 -ngl 99 -t 6, model /mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf
|
||||||
|
MAVEN_LLM_URL=http://127.0.0.1:18099 no_proxy=127.0.0.1,localhost \
|
||||||
|
deps/go/go/bin/go test -count=1 -timeout 40m \
|
||||||
|
-run TestLLMTalkBaseline ./internal/phraser/eval/ -v
|
||||||
|
```
|
||||||
|
|
||||||
|
Three runs per configuration, always. The fixture is 27 cases, so one reply
|
||||||
|
changing moves the composite by 3.7 points — a single run cannot tell a real
|
||||||
|
change from sampling noise. This was learned the expensive way: an earlier claim
|
||||||
|
that "one nudge case fails every run" turned out to be three different cases
|
||||||
|
across three runs.
|
||||||
|
|
||||||
|
**Run the box otherwise idle.** See the contamination note at the bottom.
|
||||||
|
|
||||||
|
## Composite, per configuration
|
||||||
|
|
||||||
|
| config | overall /27 | chat /9 | query /9 | knowledge /9 | canned fallbacks |
|
||||||
|
|---|---|---|---|---|---|
|
||||||
|
| baseline, no grammar | 7, 12, 7 | 1, 1, 0 | 2, 4, 2 | 4, 7, 5 | 0, 0, 0 |
|
||||||
|
| + GBNF grammar (#398) | 14, 15, 8 | 1, 3, 0 | 5, 6, 3 | 8, 6, 5 | 0, 0, 0 |
|
||||||
|
| + Russian prompts (#400) | 11, 17, 15 | 1, 5, 3 | 5, 6, 8 | 5, 6, 4 | 0, 0, 0 |
|
||||||
|
| + truncation fix, 1000ch/768tok | 12, 13, 10 | 2, 2, 1 | 7, 7, 5 | 3, 4, 4 | 3, 3, 6 |
|
||||||
|
| + rebalanced, 600ch/1024tok | **void — contaminated** | | | | |
|
||||||
|
|
||||||
|
"Canned fallbacks" counts replies that came back as the hardcoded `"не знаю."`
|
||||||
|
or `"поговорили."`. It is not a check, it is a health signal: those strings mean
|
||||||
|
the phraser gave up, and the eval scores them as ordinary bad replies.
|
||||||
|
|
||||||
|
## Per-check
|
||||||
|
|
||||||
|
| check | no grammar | + grammar | + RU prompts | + truncation fix |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| nonempty | 27, 27, 27 | 27, 27, 27 | 27, 27, 27 | 27, 27, 27 |
|
||||||
|
| ellipsis | 20, 19, 23 | 27, 27, 27 | 27, 27, 27 | 27, 27, 27 |
|
||||||
|
| lang | 13, 16, 15 | 23, 26, 26 | 25, 26, 25 | 26, 27, 27 |
|
||||||
|
| feminine | — | — | 25, 24, 26 | 25, 25, 27 |
|
||||||
|
| address | — | — | 21, 22, 22 | 22, 21, 22 |
|
||||||
|
| ontopic | — | — | 17, 24, 18 | 17, 19, 14 |
|
||||||
|
|
||||||
|
`nonempty` reading 27/27 everywhere is not good news — it was a broken check.
|
||||||
|
It tested for a non-blank string, so replies of literally `{` and `"15-16"`
|
||||||
|
passed it. Fixed on `overnight/fix-truncation`; it needs a letter now.
|
||||||
|
|
||||||
|
## What each change actually bought
|
||||||
|
|
||||||
|
**GBNF grammar (#398) — the biggest single win.** Qwen3.5-0.8B writes
|
||||||
|
`Thinking Process:` as plain text with no tags, `stripThink` only handles
|
||||||
|
`</think>`, so the JSON never closed and the plain-text fallback shipped the
|
||||||
|
literal reasoning. `ellipsis` went 20→27 and `lang` 13→26. The router had been
|
||||||
|
using a grammar for ages; the phraser asking nicely in the prompt was the
|
||||||
|
oversight.
|
||||||
|
|
||||||
|
**Russian prompts (#400) — modest, plus a large latency win.** Chat 1.3→3.0
|
||||||
|
average, query 4.7→6.3, knowledge 6.3→5.0. All inside the run-to-run spread, so
|
||||||
|
"probably better on the paths it targeted, not provable in three runs". p50
|
||||||
|
latency dropped from ~11.5s to ~2.3s and that part is consistent across all
|
||||||
|
three runs — shorter prompts, and she stopped emitting English reasoning first.
|
||||||
|
|
||||||
|
**Truncation fix — necessary, and did not help the score.** Two real bugs
|
||||||
|
(replies of `{`, and a `nonempty` check that passed them), both fixed, and the
|
||||||
|
composite went nowhere. A complete rambling wrong answer fails the same checks a
|
||||||
|
truncated one did. Worth doing anyway: the daemon was shipping `{` to a
|
||||||
|
text-to-speech voice.
|
||||||
|
|
||||||
|
## The truncation bug, since the cause was counter-intuitive
|
||||||
|
|
||||||
|
The grammar's `string ::= ... {0,400}` rule was the cause, not the token cap.
|
||||||
|
Measured against Qwen3.5-0.8B at three caps — 256, 768 and 2048 — the reply came
|
||||||
|
back **exactly 400 characters every time, cut mid-word** (`"Нужно записать и,"`).
|
||||||
|
|
||||||
|
Then I raised the bound to 1000 while the cap was 768 tokens and made it worse:
|
||||||
|
Russian runs ~1.5 characters per token here, so generation died on the *token*
|
||||||
|
cap instead, mid-object, and the new guard correctly refused it and shipped
|
||||||
|
`"не знаю."` — 3, 3 and 6 fallbacks per run, from zero. **The two limits have to
|
||||||
|
agree.** 600 characters needs ~400 tokens; the cap is 1024.
|
||||||
|
|
||||||
|
## Where the remaining failures live
|
||||||
|
|
||||||
|
`address` is stuck at 21-22 of 27 and `ontopic` at 14-19. Both resist prompting.
|
||||||
|
|
||||||
|
**The prompt now explicitly forbids exactly what she does.** It says never "вы",
|
||||||
|
use the singular — and she writes `вашей`, `подождите`, `делаете`, `хотите`,
|
||||||
|
`напишите`. Telling a 0.8B "never do X" does not work. Same for
|
||||||
|
`feminine`: `я готов`, `я понял`, `я нашел`, `я заметил`, `я сказал`.
|
||||||
|
|
||||||
|
**Some of `ontopic` is the fixture, not the model.** `chat-how-are-you` got
|
||||||
|
`"Привет! Я здесь, чтобы поговорить. Как дела сегодня?"` — a fine reply that
|
||||||
|
fails because `want_any` is `[норм, хорош, порядк, тут, работ]`. It fails in
|
||||||
|
every run, so it inflates the count. The `ontopic` column currently measures the
|
||||||
|
fixture as much as the model. Not fixed yet, deliberately: changing it would
|
||||||
|
break comparability with the runs above.
|
||||||
|
|
||||||
|
**Two replies worth reading, because they are not fixable by prompting:**
|
||||||
|
|
||||||
|
- Thunder and lightning: *"Скорость молнии — 8-10 тысяч километров в секунду, но
|
||||||
|
звук — 300 метров в секунду, что делает молнию громче."* Confidently wrong,
|
||||||
|
and it concludes lightning is *louder* rather than sound being *slower*.
|
||||||
|
- "расскажи обо мне": *"Ты — прекрасное существо, с душой и вниманием… Спасибо за
|
||||||
|
твою улыбку… О тебе — заповедь любви."* Sycophantic filler, zero information,
|
||||||
|
and precisely the "not a relationship" non-goal.
|
||||||
|
- Boiling an egg: `"15-16"` one run, `"1"` another. No unit, wrong number.
|
||||||
|
|
||||||
|
The first argues for reading instead of recalling (#403 — Kiwix retrieval scores
|
||||||
|
8/8 on the same questions given English keywords). The second and third argue
|
||||||
|
for templates on the paths where correctness matters (#392).
|
||||||
|
|
||||||
|
## Contamination note — how the last row got voided
|
||||||
|
|
||||||
|
I started the query-rewrite agent against the same llama-server the sweep was
|
||||||
|
using, and assumed contention would only affect latency. It did not. The
|
||||||
|
knowledge path collapsed to 0 of 9 with eight canned `"не знаю."` replies, p95
|
||||||
|
tripled to 23.7s, and **the report still said "0 errors"**.
|
||||||
|
|
||||||
|
That is Vikunja #397, and it is worse than filed: a merely *busy* server
|
||||||
|
produces a clean-looking report with a third of the fixture silently answering
|
||||||
|
`"не знаю."`. `PhraseChat` and `PhraseQuery` swallow every failure and return a
|
||||||
|
hardcoded string, so infrastructure trouble is indistinguishable from bad
|
||||||
|
phrasing in the score. The talk test guards the *start* and *end* of a run with
|
||||||
|
a model check, which catches a dead server but not a loaded one.
|
||||||
|
|
||||||
|
**Until #397 is fixed, treat any run made on a busy box as void.**
|
||||||
|
|
||||||
|
## Next
|
||||||
|
|
||||||
|
- Re-run 600ch/1024tok clean, to fill the void row.
|
||||||
|
- Score `Qwen3.5-2B-UD-Q4_K_XL` (already at `/mnt/hdd1/llms/qwen3.5/`, never
|
||||||
|
measured) on this fixture and the router fixture. Not the 4B — too big for
|
||||||
|
this box, owner's call.
|
||||||
|
- Newer sub-500M candidates (LFM2.5 200M/300M) are worth a run for routing.
|
||||||
|
Note `MODEL-BAKEOFF-31-07-2026.md` found LFM2.5-**1.2B** worse than
|
||||||
|
Qwen3.5-0.8B at Russian routing and 2.4× slower — but those are a different,
|
||||||
|
older generation, so that result does not predict the small ones.
|
||||||
|
- Fix `chat-how-are-you`'s `want_any`, and re-baseline once, so `ontopic`
|
||||||
|
measures the model.
|
||||||
|
- #397 first if anything, since it decides whether any of the above is
|
||||||
|
trustworthy.
|
||||||
@@ -12,7 +12,7 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
type fakeCore struct {
|
type fakeCore struct {
|
||||||
ipc.CoreAPI
|
ipc.UnimplementedCoreAPI
|
||||||
facts map[string]ipc.Fact // composite key "key|source" → Fact
|
facts map[string]ipc.Fact // composite key "key|source" → Fact
|
||||||
writeLog []ipc.WriteFactReq
|
writeLog []ipc.WriteFactReq
|
||||||
writeErr error
|
writeErr error
|
||||||
|
|||||||
@@ -0,0 +1,71 @@
|
|||||||
|
// actionTable dispatches applyAction's per-intent bodies. Each of the 7
|
||||||
|
// intents (fact, reminder, note, query, act, chat, system) has one handler
|
||||||
|
// here with the signature:
|
||||||
|
//
|
||||||
|
// func(h *reactiveHandler, ctx context.Context, dec router.Decision) string
|
||||||
|
//
|
||||||
|
// same contract as applyAction itself: "" means "let the Replier phrase the
|
||||||
|
// reply", a non-empty string OVERRIDES it. This is a straight extraction of
|
||||||
|
// applyAction's old switch cases (formerly ~300 lines in voice.go) — no
|
||||||
|
// reordering of side effects, no new abstractions inside a handler.
|
||||||
|
//
|
||||||
|
// What does NOT belong in this table, because it is not per-intent:
|
||||||
|
//
|
||||||
|
// - the dec.Clarify short-circuit ("" when the router's stage-3 fired) —
|
||||||
|
// stays in applyAction, before dispatch, since it applies to every
|
||||||
|
// intent identically.
|
||||||
|
// - the destructive-act confirm gate (park / resolveConfirm / confirmTTL)
|
||||||
|
// and the enabled-tool allowlist. Both live entirely inside
|
||||||
|
// actionAct/handleAct in actions_act.go, exactly where they lived in the old
|
||||||
|
// switch's IntentAct case — they are act-specific (a fact or a note
|
||||||
|
// can't be destructive), not shared across intents, so they do not need
|
||||||
|
// to move to a separate layer. The important invariant, preserved
|
||||||
|
// as-is: applyAction runs identically whether dec came from a fresh
|
||||||
|
// route or from a completed clarify answer (see finishClarified in
|
||||||
|
// clarify.go and its comment "filling in an argument never grants
|
||||||
|
// authority") — a handler must never special-case a clarify-completed
|
||||||
|
// decision to skip the confirm gate or the allowlist.
|
||||||
|
// - detectPattern and dialogue-session bookkeeping (rememberTurn,
|
||||||
|
// followUpMerge) run in the callers (runTurn,
|
||||||
|
// finishClarified), not per-intent, and are untouched by this slice.
|
||||||
|
//
|
||||||
|
// Each handler lives in actions_<intent>.go; the small ones (chat, system)
|
||||||
|
// and the table itself stay here.
|
||||||
|
//
|
||||||
|
// Adding an intent: write its handler in its own file, add one line to
|
||||||
|
// actionHandlers. Do not grow applyAction's switch back.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// actionHandlers is the per-intent dispatch table used by applyAction.
|
||||||
|
var actionHandlers = map[router.Intent]func(*reactiveHandler, context.Context, router.Decision) string{
|
||||||
|
router.IntentFact: (*reactiveHandler).actionFact,
|
||||||
|
router.IntentReminder: (*reactiveHandler).actionReminder,
|
||||||
|
router.IntentAct: (*reactiveHandler).actionAct,
|
||||||
|
router.IntentChat: (*reactiveHandler).actionChat,
|
||||||
|
router.IntentSystem: (*reactiveHandler).actionSystem,
|
||||||
|
router.IntentNote: (*reactiveHandler).actionNote,
|
||||||
|
router.IntentQuery: (*reactiveHandler).actionQuery,
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *reactiveHandler) actionChat(ctx context.Context, dec router.Decision) string {
|
||||||
|
// Conversational: build history from dialogue session (prior user turns)
|
||||||
|
// and let the LLM respond from general knowledge + context.
|
||||||
|
history := h.chatHistory()
|
||||||
|
reply, err := h.phraser.PhraseChat(ctx, dec.Utterance, history)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: chat: %v", err)
|
||||||
|
return "поговорили."
|
||||||
|
}
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *reactiveHandler) actionSystem(ctx context.Context, dec router.Decision) string {
|
||||||
|
return h.replySystem(ctx, dec)
|
||||||
|
}
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"log"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
"github.com/kami/maven/internal/tool"
|
||||||
|
)
|
||||||
|
|
||||||
|
// actionAct handles router.IntentAct: match a verb to an enabled tool, offer
|
||||||
|
// it to the ecosystems first, and run it behind the confirm gate and the
|
||||||
|
// allowlist. proposeGap and the confirm gate itself live in confirm.go.
|
||||||
|
func (h *reactiveHandler) actionAct(ctx context.Context, dec router.Decision) string {
|
||||||
|
// tool executor: run the matched fn against the enabled allowlist.
|
||||||
|
// HasFn=false ⇒ try the matcher (for LLM-routed acts where the verb
|
||||||
|
// didn't go through the stage-0 act grammar).
|
||||||
|
if !dec.Slots.HasFn && dec.Slots.Text != "" && h.matcher != nil {
|
||||||
|
if fn, args, ok := h.matcher.Match(dec.Slots.Text); ok {
|
||||||
|
dec.Slots.Fn, dec.Slots.Args, dec.Slots.HasFn = fn, args, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Praxis ecosystem tools: intercept before the system command executor.
|
||||||
|
if h.ecosystem != nil && h.ecosystem.praxis != nil && dec.Slots.HasFn {
|
||||||
|
if reply := h.handlePraxisAct(ctx, dec); reply != "" {
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hexis ecosystem action: if ecosystem is configured and we have a verb
|
||||||
|
// + entity text, try to resolve the entity and execute via Hexis.
|
||||||
|
if h.ecosystem != nil && h.ecosystem.hexis != nil && dec.Slots.Text != "" {
|
||||||
|
if reply := h.handleHexisAct(ctx, dec); reply != "" {
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// HasFn still false ⇒ no allowlist match: scaffold a 'proposed' tool
|
||||||
|
// the user can enable on the authed surface ("earn the right to ask").
|
||||||
|
if !dec.Slots.HasFn {
|
||||||
|
return h.proposeGap(ctx, dec)
|
||||||
|
}
|
||||||
|
out, err := h.tools.Exec(ctx, dec.Slots.Fn, dec.Slots.Args, false)
|
||||||
|
if err != nil {
|
||||||
|
switch {
|
||||||
|
case errors.Is(err, tool.ErrNeedsConfirm):
|
||||||
|
// destructive: park it and ask. The next utterance answers.
|
||||||
|
phrase := actPhrase(dec.Slots.Fn, dec.Slots.Args)
|
||||||
|
h.park(dec.Slots.Fn, dec.Slots.Args, phrase)
|
||||||
|
return "выполнить «" + phrase + "»? скажи «да» или «нет»."
|
||||||
|
case errors.Is(err, tool.ErrNotEnabled):
|
||||||
|
return h.proposeGap(ctx, dec)
|
||||||
|
}
|
||||||
|
log.Printf("voice: tool %s: %v", dec.Slots.Fn, err)
|
||||||
|
if out != "" {
|
||||||
|
return "не получилось выполнить команду: " + firstLine(out)
|
||||||
|
}
|
||||||
|
return "не получилось выполнить команду."
|
||||||
|
}
|
||||||
|
if out != "" {
|
||||||
|
return "готово: " + firstLine(out)
|
||||||
|
}
|
||||||
|
return "готово."
|
||||||
|
}
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"strconv"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// actionFact handles router.IntentFact: persist a tapped self-fact, index
|
||||||
|
// it for recall, and let pattern detection propose a routine.
|
||||||
|
func (h *reactiveHandler) actionFact(ctx context.Context, dec router.Decision) string {
|
||||||
|
if !dec.Slots.HasKey {
|
||||||
|
return "не разобрала, что записать — попробуй иначе."
|
||||||
|
}
|
||||||
|
now := h.now()
|
||||||
|
req := ipc.WriteFactReq{
|
||||||
|
Ts: now,
|
||||||
|
Kind: "self",
|
||||||
|
Key: dec.Slots.Key,
|
||||||
|
Value: dec.Slots.Value,
|
||||||
|
Source: "tap:voice",
|
||||||
|
Confidence: 1.0,
|
||||||
|
// Subject: the key doubles as the entity-resolution candidate —
|
||||||
|
// a voice-tapped fact's key is usually the thing/person it's
|
||||||
|
// about ("espresso_machine", "kate"), so queueing it for Nexus
|
||||||
|
// resolution costs one async lookup and is a no-op (not_found)
|
||||||
|
// for the abstract self-state keys (mood, water) that aren't
|
||||||
|
// entities at all.
|
||||||
|
Subject: dec.Slots.Key,
|
||||||
|
}
|
||||||
|
factID, err := h.api.WriteFact(ctx, req)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: write fact: %v", err)
|
||||||
|
return "не получилось сохранить факт."
|
||||||
|
}
|
||||||
|
// Index the fact utterance in long-term memory (best-effort, must not
|
||||||
|
// fail the fact write). Facts aren't in the notes table, so this is the
|
||||||
|
// only recall path for them — "когда я пил воду?" reads back from here.
|
||||||
|
if h.memStore != nil {
|
||||||
|
if vec, err := router.EmbedPassage(ctx, h.embedder, dec.Utterance); err != nil {
|
||||||
|
log.Printf("voice: embed fact for memory: %v", err)
|
||||||
|
} else if err := h.memStore.Insert(ctx, "fact:"+dec.Slots.Key+":"+strconv.FormatInt(now.Unix(), 10), vec, map[string]string{
|
||||||
|
"source": "voice",
|
||||||
|
"type": "fact",
|
||||||
|
"text": dec.Utterance,
|
||||||
|
"ts": strconv.FormatInt(now.Unix(), 10),
|
||||||
|
}); err != nil {
|
||||||
|
log.Printf("voice: memory insert fact: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Event extraction + pattern detection (best-effort, must not fail the
|
||||||
|
// fact write). If the fact describes a recognizable action, it becomes a
|
||||||
|
// normalized event; if ≥3 events for the same action+object show stable
|
||||||
|
// intervals, a proposed routine is created and parked for confirmation.
|
||||||
|
if h.dataStore != nil {
|
||||||
|
if phrase := h.detectPattern(ctx, factID, dec.Slots.Key, dec.Slots.Value, now); phrase != "" {
|
||||||
|
return phrase // "ты заправляешь ... напоминать?"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "" // replier phrases the success reply
|
||||||
|
}
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"strconv"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// actionNote handles router.IntentNote: embed the note, persist it, and
|
||||||
|
// index it for recall.
|
||||||
|
func (h *reactiveHandler) actionNote(ctx context.Context, dec router.Decision) string {
|
||||||
|
// embed the note text with the same model the classifier uses, persist
|
||||||
|
// via CoreAPI (source=tap:voice). Semantic recall lives in `notes`, not
|
||||||
|
// facts — no predicate reads it (spec's two-memory split).
|
||||||
|
vec, err := router.EmbedPassage(ctx, h.embedder, dec.Utterance)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: embed note: %v", err)
|
||||||
|
return "не получилось сохранить заметку."
|
||||||
|
}
|
||||||
|
noteTs := h.now()
|
||||||
|
noteID, err := h.api.WriteNote(ctx, noteTs, dec.Utterance, vec, "tap:voice")
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: write note: %v", err)
|
||||||
|
return "не получилось сохранить заметку."
|
||||||
|
}
|
||||||
|
// Insert into long-term memory (best-effort, must not fail the note write).
|
||||||
|
// text/ts in the meta make a Search hit self-describing (see bestRecall).
|
||||||
|
if h.memStore != nil {
|
||||||
|
if err := h.memStore.Insert(ctx, "note:"+strconv.FormatInt(noteID, 10), vec, map[string]string{
|
||||||
|
"source": "voice",
|
||||||
|
"type": "note",
|
||||||
|
"text": dec.Utterance,
|
||||||
|
"ts": strconv.FormatInt(noteTs.Unix(), 10),
|
||||||
|
}); err != nil {
|
||||||
|
log.Printf("voice: memory insert: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "" // replier phrases the "saved" reply
|
||||||
|
}
|
||||||
@@ -0,0 +1,226 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/memory"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
"github.com/kami/maven/internal/weather"
|
||||||
|
)
|
||||||
|
|
||||||
|
// queryTurn is the per-turn scratch a chain of query sources shares: the
|
||||||
|
// decision being answered plus the work an earlier source already paid for
|
||||||
|
// (the query embedding, the notes it pulled). Sources read and fill it in
|
||||||
|
// order, so a later source never re-embeds.
|
||||||
|
type queryTurn struct {
|
||||||
|
dec router.Decision
|
||||||
|
vec []float32
|
||||||
|
notes []ipc.Note
|
||||||
|
}
|
||||||
|
|
||||||
|
// querySource — one answer source in the chain actionQuery walks. answer
|
||||||
|
// returns (reply, true) when this source claims the question, ("", false)
|
||||||
|
// when it passes to the next one. name is for reading the table, not logged.
|
||||||
|
//
|
||||||
|
// A struct of one func rather than an interface: every source is a plain
|
||||||
|
// method on *reactiveHandler with no state of its own (what state a turn has
|
||||||
|
// lives in queryTurn), so an interface would mean one empty type per source
|
||||||
|
// to satisfy it — ceremony for nothing. Same reasoning as confirmResolver in
|
||||||
|
// confirm.go, and the table then reads like actionHandlers: a flat list of
|
||||||
|
// method expressions you extend with one line.
|
||||||
|
type querySource struct {
|
||||||
|
name string
|
||||||
|
answer func(*reactiveHandler, context.Context, *queryTurn) (string, bool)
|
||||||
|
}
|
||||||
|
|
||||||
|
// querySources is the ordered chain actionQuery walks; first source to claim
|
||||||
|
// answers the turn. THE ORDER IS LOAD-BEARING — see the memory-before-notes
|
||||||
|
// comment on queryMemory: running the notes-only pass first was #373, and the
|
||||||
|
// gate was never the bug. Adding a source (Kiwix, RSS, crawler, email) is one
|
||||||
|
// line here plus its method; where you put the line is the whole decision.
|
||||||
|
var querySources = []querySource{
|
||||||
|
{"fact-by-key", (*reactiveHandler).queryFactByKey},
|
||||||
|
{"calendar", (*reactiveHandler).queryCalendar},
|
||||||
|
{"weather", (*reactiveHandler).queryWeather},
|
||||||
|
{"embed", (*reactiveHandler).queryEmbed},
|
||||||
|
{"memory", (*reactiveHandler).queryMemory},
|
||||||
|
{"notes", (*reactiveHandler).queryNotes},
|
||||||
|
{"general-knowledge", (*reactiveHandler).queryGeneral},
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *reactiveHandler) actionQuery(ctx context.Context, dec router.Decision) string {
|
||||||
|
t := &queryTurn{dec: dec}
|
||||||
|
for _, src := range querySources {
|
||||||
|
if reply, ok := src.answer(h, ctx, t); ok {
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "не знаю."
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryFactByKey — when the dialogue layer resolved an anaphoric reference to
|
||||||
|
// a prior fact's key (e.g. "когда я это сделал?" after "запиши что я пил
|
||||||
|
// воду"), look up the fact's value directly.
|
||||||
|
func (h *reactiveHandler) queryFactByKey(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
dec := t.dec
|
||||||
|
if !dec.Slots.HasKey || dec.Slots.Key == "" {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
f, err := h.api.LatestFact(ctx, dec.Slots.Key)
|
||||||
|
if err != nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
if dec.Slots.HasTime {
|
||||||
|
// The query asks about timing — the fact's own timestamp is the
|
||||||
|
// answer it's looking for. Format as a natural reply.
|
||||||
|
return fmt.Sprintf("я записала это %s", formatTime(f.Ts)), true
|
||||||
|
}
|
||||||
|
// General fact reference: describe what we know.
|
||||||
|
if dec.Utterance == "" {
|
||||||
|
return fmt.Sprintf("вот что я знаю: %s — %s", dec.Slots.Key, f.Value), true
|
||||||
|
}
|
||||||
|
// The utterance still carries the question; fall through to normal RAG
|
||||||
|
// with the resolved key in context.
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryCalendar — "что у меня сегодня?", "планы на завтра?"
|
||||||
|
// h.now(), not time.Now(): the handler's clock is the injected one, so this
|
||||||
|
// source can be tested at a fixed time like the rest.
|
||||||
|
func (h *reactiveHandler) queryCalendar(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
date, ok := router.ParseCalendarDate(t.dec.Utterance, h.now())
|
||||||
|
if !ok {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
events, err := h.api.CalendarEvents(ctx, date, date.Add(24*time.Hour))
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: calendar events: %v", err)
|
||||||
|
return "не получилось проверить календарь.", true
|
||||||
|
}
|
||||||
|
values := make([]string, len(events))
|
||||||
|
for i, e := range events {
|
||||||
|
values[i] = e.Value
|
||||||
|
}
|
||||||
|
var f router.CalendarEventFormatter
|
||||||
|
return f.Format(values, date), true
|
||||||
|
}
|
||||||
|
|
||||||
|
func (h *reactiveHandler) queryWeather(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
if !isWeatherQuery(t.dec.Utterance) {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
loc := extractWeatherLocation(t.dec.Utterance, h.weatherLocation)
|
||||||
|
ctxWT, cancel := context.WithTimeout(ctx, 5*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
w, err := h.weatherProvider.CurrentWeather(ctxWT, loc)
|
||||||
|
if errors.Is(err, weather.ErrNotConfigured) {
|
||||||
|
return "погода не настроена.", true
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: weather: %v", err)
|
||||||
|
return "не получилось узнать погоду.", true
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("в %s сейчас %.0f градусов, %s.", w.Location, w.Temperature, w.Condition), true
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryEmbed isn't an answer source — it's the shared cost the two recall
|
||||||
|
// sources below both need, run once, in the position it always ran in. It
|
||||||
|
// only claims the turn when the embedder fails.
|
||||||
|
func (h *reactiveHandler) queryEmbed(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
vec, err := router.EmbedQuery(ctx, h.embedder, t.dec.Utterance)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: embed query: %v", err)
|
||||||
|
return "не получилось найти ответ.", true
|
||||||
|
}
|
||||||
|
t.vec = vec
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryMemory — long-term memory first: ONE search over everything Maven
|
||||||
|
// remembers (notes and facts share this index) and ONE confidence gate, so
|
||||||
|
// the memory that is clearly the best match answers — a note just as much as
|
||||||
|
// a fact.
|
||||||
|
//
|
||||||
|
// This used to run only after the notes-only source below had already
|
||||||
|
// rejected the same note at the same score, which no note could ever survive
|
||||||
|
// a second time: the branch could only return a fact (#373). Order, not the
|
||||||
|
// gate, was the bug — the set of questions Maven answers is unchanged, only
|
||||||
|
// which memory gets to answer them.
|
||||||
|
func (h *reactiveHandler) queryMemory(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
if h.memStore == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
hits, herr := h.memStore.Search(ctx, t.vec, 3)
|
||||||
|
if herr != nil {
|
||||||
|
log.Printf("voice: memory search: %v", herr)
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
hit, ok := bestRecall(hits, h.queryMinScore, h.queryMinMargin)
|
||||||
|
if !ok {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
text := hit.Meta["text"]
|
||||||
|
// A note is phrased in Maven's voice; a fact is read back as it was
|
||||||
|
// stored.
|
||||||
|
if hit.Meta["type"] == "note" {
|
||||||
|
if reply, perr := h.phraser.PhraseQuery(ctx, t.dec.Utterance, []string{text}); perr == nil && reply != "" {
|
||||||
|
return reply, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return text, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryNotes — notes-only pass, for notes the vector index above does not
|
||||||
|
// hold (an older note written before it existed). Same gate, notes-only
|
||||||
|
// candidates.
|
||||||
|
//
|
||||||
|
// Confidence gate: below it, say "I don't know" rather than read back the
|
||||||
|
// least-unrelated note — a confident wrong recall is worse than a gap (spec's
|
||||||
|
// "not a guesser-of-truth"). Same instinct as the loop's since(key)==null →
|
||||||
|
// don't fire. Two parts: an absolute cosine floor, and a margin over the
|
||||||
|
// runner-up, which is the part that works with the e5 embedder's narrow score
|
||||||
|
// band. See memory.Confident. Failing the gate passes the turn on to general
|
||||||
|
// knowledge, which is what "don't read back the runner-up" means here.
|
||||||
|
func (h *reactiveHandler) queryNotes(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
notes, err := h.api.QueryNotes(ctx, t.vec, 5)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: query notes: %v", err)
|
||||||
|
return "не получилось найти ответ.", true
|
||||||
|
}
|
||||||
|
t.notes = notes
|
||||||
|
noteScores := make([]float64, len(notes))
|
||||||
|
for i, n := range notes {
|
||||||
|
noteScores[i] = n.Score
|
||||||
|
}
|
||||||
|
if !memory.ConfidentScores(noteScores, h.queryMinScore, h.queryMinMargin) {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
texts := make([]string, len(notes))
|
||||||
|
for i, n := range notes {
|
||||||
|
texts[i] = n.Text
|
||||||
|
}
|
||||||
|
reply, err := h.phraser.PhraseQuery(ctx, t.dec.Utterance, texts)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: phrase query: %v", err)
|
||||||
|
}
|
||||||
|
if reply == "" {
|
||||||
|
reply = "вот что я нашла: " + texts[0]
|
||||||
|
}
|
||||||
|
return reply, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryGeneral — general knowledge from the phraser, the last source before
|
||||||
|
// giving up. It always claims: either the model answers or Maven says she
|
||||||
|
// doesn't know.
|
||||||
|
func (h *reactiveHandler) queryGeneral(ctx context.Context, t *queryTurn) (string, bool) {
|
||||||
|
reply, err := h.phraser.PhraseQuery(ctx, t.dec.Utterance, nil)
|
||||||
|
if err != nil || reply == "" {
|
||||||
|
return "не знаю.", true
|
||||||
|
}
|
||||||
|
return reply, true
|
||||||
|
}
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// actionReminder handles router.IntentReminder: parse the time when stage-0
|
||||||
|
// skipped the extractor, then create the reminder.
|
||||||
|
func (h *reactiveHandler) actionReminder(ctx context.Context, dec router.Decision) string {
|
||||||
|
if !dec.Slots.HasTime {
|
||||||
|
// Stage-0 (reminder-wakeword grammar) skips the extractor, so the
|
||||||
|
// time wasn't parsed. Run the parser as a fallback.
|
||||||
|
if dec.Stage == 0 && h.timeParser != nil {
|
||||||
|
t, ok, err := h.timeParser.Parse(ctx, dec.Utterance, h.now())
|
||||||
|
if err == nil && ok {
|
||||||
|
dec.Slots.Time = t
|
||||||
|
dec.Slots.HasTime = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !dec.Slots.HasTime {
|
||||||
|
return "не получилось разобрать время напоминания."
|
||||||
|
}
|
||||||
|
}
|
||||||
|
payload := `{"text":` + jsonString(dec.Utterance) + `}`
|
||||||
|
if _, err := h.api.CreateReminder(ctx, dec.Slots.Time, payload, ""); err != nil {
|
||||||
|
log.Printf("voice: create reminder: %v", err)
|
||||||
|
return "не получилось поставить напоминание."
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
@@ -0,0 +1,292 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"math/rand"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/dialogue"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// clarifyTTL — how long a parked question stays answerable. Same 90s as the
|
||||||
|
// confirm gate, for the same reason: an answer is a same-breath gesture, and a
|
||||||
|
// stale question must not eat an unrelated later utterance.
|
||||||
|
const clarifyTTL = 90 * time.Second
|
||||||
|
|
||||||
|
// wantedSlots — what each intent needs before she can act on it. First entry is
|
||||||
|
// the one she asks about; the rest are only used to decide act-vs-drop.
|
||||||
|
//
|
||||||
|
// Intents not listed here are never worth a question: note and query act on the
|
||||||
|
// raw utterance, chat and system have nothing to fill in. For those a clarify
|
||||||
|
// decision keeps the canned "не поняла" reply — inventing a question for noise
|
||||||
|
// is worse than admitting she missed it.
|
||||||
|
// A reminder wants BOTH what to remind about and when. Subject first: "напомни
|
||||||
|
// в 11" has a time and nothing to say at 11, and a reminder with no subject is
|
||||||
|
// not worth setting. Order here is the order she asks in — she still only asks
|
||||||
|
// about the first one missing.
|
||||||
|
var wantedSlots = map[router.Intent][]dialogue.Slot{
|
||||||
|
router.IntentReminder: {dialogue.SlotText, dialogue.SlotTime},
|
||||||
|
router.IntentFact: {dialogue.SlotKey},
|
||||||
|
router.IntentAct: {dialogue.SlotFn},
|
||||||
|
}
|
||||||
|
|
||||||
|
// clarifyQuestions — one short question per missing slot.
|
||||||
|
//
|
||||||
|
// These are fixed templates, not model output. The resident model is a 0.8B; it
|
||||||
|
// would wander, and a question whose wording changes every time is harder to
|
||||||
|
// answer than a blunt one that always reads the same. They are infinitive
|
||||||
|
// questions, so there is no gender agreement to get wrong; the feminine
|
||||||
|
// self-reference lives in the reply she gives when she drops the request.
|
||||||
|
var clarifyQuestions = map[dialogue.Slot]string{
|
||||||
|
dialogue.SlotTime: "Когда?",
|
||||||
|
dialogue.SlotText: "О чём напомнить?",
|
||||||
|
dialogue.SlotKey: "Что записать?",
|
||||||
|
dialogue.SlotFn: "Что сделать?",
|
||||||
|
}
|
||||||
|
|
||||||
|
// clarifyGaveUp — she is out of questions and still does not have the slot. She
|
||||||
|
// says so out loud: dropping the request in silence would leave him thinking it
|
||||||
|
// landed. Feminine self-reference ("поняла"), as everywhere.
|
||||||
|
const clarifyGaveUp = "Прости, я не поняла. Скажи, пожалуйста, по-другому."
|
||||||
|
|
||||||
|
// clarifyExpiredVariants — his answer came after the TTL, so the parked request
|
||||||
|
// is already gone. Same tone as clarifyGaveUp, different reason: too much time
|
||||||
|
// passed, not "I did not understand". Feminine self-reference ("ждала",
|
||||||
|
// "отпустила"); he is addressed with a plain imperative.
|
||||||
|
//
|
||||||
|
// Five phrasings, not one. This is the line he hears whenever he walks off
|
||||||
|
// mid-request, so it is the line that repeats most — and the same sentence every
|
||||||
|
// time is what makes a house assistant sound like a kiosk. They all carry the
|
||||||
|
// same two facts (the old request is gone; say it again if it still matters),
|
||||||
|
// because the wording may vary and the meaning may not.
|
||||||
|
//
|
||||||
|
// Fixed templates rather than model output, for the same reason as
|
||||||
|
// clarifyQuestions: this text has to be right every time, and it is not worth a
|
||||||
|
// generation to say something this small.
|
||||||
|
var clarifyExpiredVariants = []string{
|
||||||
|
"Прости, я слишком долго ждала ответа и отпустила прошлую просьбу. Если она ещё нужна, скажи заново.",
|
||||||
|
"Кажется, прошлая просьба уже не важна — я её отпустила. Если я ошибаюсь, повтори.",
|
||||||
|
"Ты как-то резко замолчал, и я не стала ждать дальше. Если та просьба ещё нужна, скажи заново.",
|
||||||
|
"Я не дождалась ответа и убрала прошлую просьбу. Повтори, если она всё ещё нужна.",
|
||||||
|
"Столько времени прошло, что я отпустила прошлую просьбу. Скажи заново, если она в силе.",
|
||||||
|
}
|
||||||
|
|
||||||
|
// clarifyExpiredLine picks one of them at random.
|
||||||
|
func clarifyExpiredLine() string {
|
||||||
|
return clarifyExpiredVariants[rand.Intn(len(clarifyExpiredVariants))]
|
||||||
|
}
|
||||||
|
|
||||||
|
// isClarifyExpired reports whether s opens with any of the expiry lines. The
|
||||||
|
// notice is glued in front of this turn's reply (see withNotice), so a caller
|
||||||
|
// checking for it has to match a prefix, not the whole string.
|
||||||
|
func isClarifyExpired(s string) bool {
|
||||||
|
for _, v := range clarifyExpiredVariants {
|
||||||
|
if strings.HasPrefix(s, v) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// trimClarifyExpired strips a leading expiry notice, leaving this turn's actual
|
||||||
|
// reply. "" ⇒ the notice was the whole thing.
|
||||||
|
func trimClarifyExpired(s string) string {
|
||||||
|
for _, v := range clarifyExpiredVariants {
|
||||||
|
if strings.HasPrefix(s, v) {
|
||||||
|
return strings.TrimSpace(strings.TrimPrefix(s, v))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(s)
|
||||||
|
}
|
||||||
|
|
||||||
|
// clarifyExpiredNotice returns that line when a parked question had just timed
|
||||||
|
// out, and "" when nothing was parked. Call it right after
|
||||||
|
// resolveClarifyAnswer: a live question is answered there, an expired one is
|
||||||
|
// only reported here — the words themselves still go on to be routed fresh.
|
||||||
|
func (h *reactiveHandler) clarifyExpiredNotice() string {
|
||||||
|
if h.clarifyStore == nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
if !h.clarifyStore.TakeExpired(voiceDialogueID, h.now()) {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
log.Printf("voice: clarify — parked question expired, telling him and routing the words fresh")
|
||||||
|
return clarifyExpiredLine()
|
||||||
|
}
|
||||||
|
|
||||||
|
// withNotice glues the expiry notice in front of this turn's reply. One turn
|
||||||
|
// carries one reply on the wire, so the notice cannot be a message of its own —
|
||||||
|
// but neither the notice nor the fresh answer may be dropped.
|
||||||
|
func withNotice(notice, reply string) string {
|
||||||
|
if notice == "" {
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
if reply == "" {
|
||||||
|
return notice
|
||||||
|
}
|
||||||
|
return notice + " " + reply
|
||||||
|
}
|
||||||
|
|
||||||
|
// missingFor returns the slots a decision still needs, most important first.
|
||||||
|
// Empty ⇒ there is nothing identifiable to ask about.
|
||||||
|
func missingFor(dec router.Decision) []dialogue.Slot {
|
||||||
|
return dialogue.StillMissing(wantedSlots[dec.Intent], toDialogueSlots(dec.Slots))
|
||||||
|
}
|
||||||
|
|
||||||
|
// clarifyQuestion picks the one question to ask for a clarify decision. Returns
|
||||||
|
// ("", false) when she has no idea what is missing.
|
||||||
|
//
|
||||||
|
// One question about one thing: if two slots are missing she asks about the
|
||||||
|
// first and lets the rest go. Two questions in a row is an interrogation.
|
||||||
|
func clarifyQuestion(dec router.Decision) (dialogue.Slot, string, bool) {
|
||||||
|
missing := missingFor(dec)
|
||||||
|
if len(missing) == 0 {
|
||||||
|
return "", "", false
|
||||||
|
}
|
||||||
|
q, ok := clarifyQuestions[missing[0]]
|
||||||
|
if !ok {
|
||||||
|
return "", "", false
|
||||||
|
}
|
||||||
|
return missing[0], q, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// askClarify parks the request and returns the question to ask instead of the
|
||||||
|
// canned "не поняла". Returns ("", false) when there is nothing to ask about, so
|
||||||
|
// the caller falls back to the canned reply.
|
||||||
|
func (h *reactiveHandler) askClarify(dec router.Decision) (string, bool) {
|
||||||
|
if h.clarifyStore == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
slot, question, ok := clarifyQuestion(dec)
|
||||||
|
if !ok {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
h.clarifyStore.Put(voiceDialogueID, &dialogue.PendingQuestion{
|
||||||
|
Intent: dialogue.Intent(dec.Intent),
|
||||||
|
Slots: toDialogueSlots(dec.Slots),
|
||||||
|
Missing: []dialogue.Slot{slot},
|
||||||
|
Utterance: dec.Utterance,
|
||||||
|
Asked: h.now(),
|
||||||
|
TTL: clarifyTTL,
|
||||||
|
Attempts: 1, // this ask
|
||||||
|
MaxAttempts: h.clarifyMaxAttempts,
|
||||||
|
})
|
||||||
|
log.Printf("voice: clarify — asked about %s for intent=%s", slot, dec.Intent)
|
||||||
|
return question, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveClarifyAnswer reads an utterance as the answer to a parked question.
|
||||||
|
// Returns ("", false) when no live question is parked (or it expired), so the
|
||||||
|
// caller routes the utterance normally as a fresh request. Sibling of
|
||||||
|
// resolveConfirm and checked in the same place.
|
||||||
|
//
|
||||||
|
// The answer is parsed with the same extractor the router uses, for the intent
|
||||||
|
// she parked — no second parser. If it still does not fill the gap she asks
|
||||||
|
// again, up to MaxAttempts; after that she says out loud that she did not
|
||||||
|
// understand. She never drops the request in silence.
|
||||||
|
func (h *reactiveHandler) resolveClarifyAnswer(ctx context.Context, text string) (string, bool) {
|
||||||
|
if h.clarifyStore == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
q := h.clarifyStore.Get(voiceDialogueID, h.now())
|
||||||
|
if q == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
intent := router.Intent(q.Intent)
|
||||||
|
answer := h.extractor.Extract(ctx, intent, text, h.now())
|
||||||
|
merged := q.Answer(text, toDialogueSlots(answer))
|
||||||
|
if len(dialogue.StillMissing(q.Missing, merged)) > 0 {
|
||||||
|
return h.reaskOrGiveUp(q, merged, text), true
|
||||||
|
}
|
||||||
|
h.clarifyStore.Delete(voiceDialogueID)
|
||||||
|
|
||||||
|
// Rebuild the decision as if it had routed cleanly, then run it down the
|
||||||
|
// normal path. Clarify is deliberately false and the intent is unchanged:
|
||||||
|
// filling in an argument never grants authority, so the completed decision
|
||||||
|
// still meets the allowlist and the destructive-act confirm gate in
|
||||||
|
// applyAction exactly like any other decision.
|
||||||
|
dec := router.Decision{
|
||||||
|
Utterance: q.Utterance,
|
||||||
|
Stage: 2,
|
||||||
|
Intent: intent,
|
||||||
|
Slots: applyDialogueSlots(answer, merged),
|
||||||
|
}
|
||||||
|
return h.finishClarified(ctx, dec), true
|
||||||
|
}
|
||||||
|
|
||||||
|
// reaskOrGiveUp handles an answer that left the gap open: ask the same question
|
||||||
|
// again while she has attempts left, otherwise say she did not understand and
|
||||||
|
// let the request go. Never returns "" — a mute give-up reads as "done".
|
||||||
|
func (h *reactiveHandler) reaskOrGiveUp(q *dialogue.PendingQuestion, merged dialogue.Slots, text string) string {
|
||||||
|
question := ""
|
||||||
|
if len(q.Missing) > 0 {
|
||||||
|
question = clarifyQuestions[q.Missing[0]]
|
||||||
|
}
|
||||||
|
if question == "" || !q.CanAsk() {
|
||||||
|
h.clarifyStore.Delete(voiceDialogueID)
|
||||||
|
log.Printf("voice: clarify — gave up on %v after %d question(s), answer was %q", q.Missing, q.Attempts, text)
|
||||||
|
return clarifyGaveUp
|
||||||
|
}
|
||||||
|
// Re-park with whatever the answer DID give, the clock restarted and one
|
||||||
|
// more question spent.
|
||||||
|
q.Slots = merged
|
||||||
|
q.Attempts++
|
||||||
|
q.Asked = h.now()
|
||||||
|
h.clarifyStore.Put(voiceDialogueID, q)
|
||||||
|
log.Printf("voice: clarify — answer %q did not fill %v, asking again (attempt %d)", text, q.Missing, q.Attempts)
|
||||||
|
return question
|
||||||
|
}
|
||||||
|
|
||||||
|
// finishClarified runs a completed decision through the same steps a freshly
|
||||||
|
// routed one takes: remember the turn, act, then phrase.
|
||||||
|
func (h *reactiveHandler) finishClarified(ctx context.Context, dec router.Decision) string {
|
||||||
|
if h.dialogueSessions != nil {
|
||||||
|
now := h.now()
|
||||||
|
prev := h.dialogueSessions.Get(voiceDialogueID, now)
|
||||||
|
dec = followUpMerge(prev, dec, now)
|
||||||
|
h.rememberTurn(prev, dec, now)
|
||||||
|
}
|
||||||
|
reply := h.applyAction(ctx, dec)
|
||||||
|
if reply == "" {
|
||||||
|
reply = h.replier.Reply(dec)
|
||||||
|
}
|
||||||
|
if reply == "" {
|
||||||
|
// Belt: an empty reply here would be a silent drop.
|
||||||
|
reply = clarifyGaveUp
|
||||||
|
}
|
||||||
|
return reply
|
||||||
|
}
|
||||||
|
|
||||||
|
// rememberTurn stores this turn as the dialogue session the next follow-up
|
||||||
|
// inherits from, carrying up to 4 prior turns of history for anaphora. Capped so
|
||||||
|
// one long conversation can't grow the session unboundedly.
|
||||||
|
func (h *reactiveHandler) rememberTurn(prev *dialogue.Session, dec router.Decision, now time.Time) {
|
||||||
|
var history []dialogue.Turn
|
||||||
|
if prev != nil {
|
||||||
|
history = append(history, dialogue.Turn{
|
||||||
|
Intent: prev.Intent,
|
||||||
|
Slots: prev.Slots,
|
||||||
|
Text: prev.Slots.Text,
|
||||||
|
})
|
||||||
|
maxHist := len(prev.History)
|
||||||
|
if maxHist > 3 {
|
||||||
|
maxHist = 3
|
||||||
|
}
|
||||||
|
history = append(history, prev.History[:maxHist]...)
|
||||||
|
}
|
||||||
|
ttl := time.Duration(0) // use the store default (2 min)
|
||||||
|
if dec.Intent == router.IntentChat {
|
||||||
|
ttl = 15 * time.Minute // conversational turns should last longer
|
||||||
|
}
|
||||||
|
h.dialogueSessions.Put(voiceDialogueID, &dialogue.Session{
|
||||||
|
Intent: dialogue.Intent(dec.Intent),
|
||||||
|
Slots: toDialogueSlots(dec.Slots),
|
||||||
|
Timestamp: now,
|
||||||
|
TTL: ttl,
|
||||||
|
History: history,
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -0,0 +1,332 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/dialogue"
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
"github.com/kami/maven/internal/tool"
|
||||||
|
"github.com/kami/maven/internal/voice"
|
||||||
|
)
|
||||||
|
|
||||||
|
// newClarifyHandler builds a handler with the clarify path wired and no model:
|
||||||
|
// stub date parser, the real fact parser, and a matcher over whatever tools the
|
||||||
|
// test enabled. `now` is fixed so TTL behaviour is testable.
|
||||||
|
func newClarifyHandler(t *testing.T) (*reactiveHandler, *store.Store, *time.Time) {
|
||||||
|
t.Helper()
|
||||||
|
st := newTestStore(t)
|
||||||
|
api := ipc.NewStoreAPI(st)
|
||||||
|
now := time.Date(2026, 7, 31, 9, 0, 0, 0, time.UTC)
|
||||||
|
matcher := tool.NewMatcher(api)
|
||||||
|
h := &reactiveHandler{
|
||||||
|
api: api,
|
||||||
|
dataStore: st,
|
||||||
|
tools: tool.NewExecutor(api, 2*time.Second),
|
||||||
|
matcher: matcher,
|
||||||
|
replier: voice.NewStubReplier(),
|
||||||
|
now: func() time.Time { return now },
|
||||||
|
dialogueSessions: dialogue.NewSessionStore(2 * time.Minute),
|
||||||
|
clarifyStore: dialogue.NewClarifyStore(clarifyTTL),
|
||||||
|
extractor: router.Extractor{
|
||||||
|
Time: router.StubDateTimeParser{},
|
||||||
|
Acts: matcher,
|
||||||
|
Facts: router.DefaultFactParser{},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
return h, st, &now
|
||||||
|
}
|
||||||
|
|
||||||
|
func clarifyDec(intent router.Intent, slots router.Slots, utterance string) router.Decision {
|
||||||
|
return router.Decision{Utterance: utterance, Stage: 3, Intent: intent, Slots: slots, Clarify: true}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyQuestionForMissingSlot pins which question goes with which gap, and
|
||||||
|
// which intents get no question at all.
|
||||||
|
func TestClarifyQuestionForMissingSlot(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
dec router.Decision
|
||||||
|
want string
|
||||||
|
asked bool
|
||||||
|
}{
|
||||||
|
{"reminder without a time", clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме"), "Когда?", true},
|
||||||
|
{"fact without a key", clarifyDec(router.IntentFact, router.Slots{Text: "запиши"}, "запиши"), "Что записать?", true},
|
||||||
|
{"act without a fn", clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это"), "Что сделать?", true},
|
||||||
|
// A time with nothing to say at that time is still half a reminder, so
|
||||||
|
// the subject is what she asks about — not silence.
|
||||||
|
{"reminder that has a time but no subject", clarifyDec(router.IntentReminder, router.Slots{HasTime: true}, "напомни в 11"), "О чём напомнить?", true},
|
||||||
|
{"reminder that has both", clarifyDec(router.IntentReminder, router.Slots{Text: "позвонить маме", HasTime: true}, "напомни в 11 позвонить маме"), "", false},
|
||||||
|
{"chat is never worth a question", clarifyDec(router.IntentChat, router.Slots{Text: "мгм"}, "мгм"), "", false},
|
||||||
|
{"query is never worth a question", clarifyDec(router.IntentQuery, router.Slots{Text: "а"}, "а"), "", false},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
_, got, asked := clarifyQuestion(tc.dec)
|
||||||
|
if asked != tc.asked || got != tc.want {
|
||||||
|
t.Errorf("%s: got (%q, %v), want (%q, %v)", tc.name, got, asked, tc.want, tc.asked)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyReminderCompletesOnAnswer is the whole point of the feature: she
|
||||||
|
// asks for the missing time and the answer creates the reminder.
|
||||||
|
func TestClarifyReminderCompletesOnAnswer(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
|
||||||
|
question, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме"))
|
||||||
|
if !asked || question != "Когда?" {
|
||||||
|
t.Fatalf("expected the time question, got %q asked=%v", question, asked)
|
||||||
|
}
|
||||||
|
|
||||||
|
reply, handled := h.resolveClarifyAnswer(ctx, "в 11:00")
|
||||||
|
if !handled {
|
||||||
|
t.Fatal("the answer to an open question must be consumed as an answer")
|
||||||
|
}
|
||||||
|
if reply == clarifyGaveUp {
|
||||||
|
t.Fatalf("a good answer must not drop the request: %q", reply)
|
||||||
|
}
|
||||||
|
|
||||||
|
reminders, err := st.DueReminders(ctx, h.now().Add(48*time.Hour))
|
||||||
|
if err != nil || len(reminders) != 1 {
|
||||||
|
t.Fatalf("clarified reminder was not created: reminders=%v err=%v", reminders, err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(reminders[0].Payload, "маме") {
|
||||||
|
t.Fatalf("the reminder lost the original request: %q", reminders[0].Payload)
|
||||||
|
}
|
||||||
|
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||||
|
t.Fatal("the question must be cleared once answered")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyFactCompletesOnAnswer — the fact path, where the answer carries
|
||||||
|
// both the key and the value.
|
||||||
|
func TestClarifyFactCompletesOnAnswer(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentFact, router.Slots{Text: "запиши"}, "запиши")); !asked {
|
||||||
|
t.Fatal("a fact with no key should be asked about")
|
||||||
|
}
|
||||||
|
if reply, handled := h.resolveClarifyAnswer(ctx, "пил воду"); !handled || reply == clarifyGaveUp {
|
||||||
|
t.Fatalf("answer should complete the fact, handled=%v reply=%q", handled, reply)
|
||||||
|
}
|
||||||
|
if fact, err := st.LatestFact(ctx, "water"); err != nil || fact.Key != "water" {
|
||||||
|
t.Fatalf("clarified fact was not written: fact=%+v err=%v", fact, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyAnswerAfterTTLIsANewRequest — a late answer is not an answer.
|
||||||
|
func TestClarifyAnswerAfterTTLIsANewRequest(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, now := newClarifyHandler(t)
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||||
|
t.Fatal("expected a question")
|
||||||
|
}
|
||||||
|
*now = now.Add(clarifyTTL + time.Second)
|
||||||
|
|
||||||
|
if reply, handled := h.resolveClarifyAnswer(ctx, "в 11:00"); handled {
|
||||||
|
t.Fatalf("an answer past the TTL must fall through to normal routing, got %q", reply)
|
||||||
|
}
|
||||||
|
if reminders, err := st.DueReminders(ctx, now.Add(48*time.Hour)); err != nil || len(reminders) != 0 {
|
||||||
|
t.Fatalf("expired question must not create anything: reminders=%v err=%v", reminders, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyAsksThreeTimesThenSaysSo — three questions are allowed, the fourth
|
||||||
|
// is not, and running out is SPOKEN. Silence would read as "handled".
|
||||||
|
func TestClarifyAsksThreeTimesThenSaysSo(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||||
|
t.Fatal("expected a first question")
|
||||||
|
}
|
||||||
|
// Two more unclear answers ⇒ two more questions (3 asks in total).
|
||||||
|
for i := 2; i <= 3; i++ {
|
||||||
|
reply, handled := h.resolveClarifyAnswer(ctx, "ну не знаю")
|
||||||
|
if !handled {
|
||||||
|
t.Fatalf("answer %d must be consumed as an answer", i)
|
||||||
|
}
|
||||||
|
if reply != "Когда?" {
|
||||||
|
t.Fatalf("attempt %d should ask again, got %q", i, reply)
|
||||||
|
}
|
||||||
|
if h.clarifyStore.Get(voiceDialogueID, h.now()) == nil {
|
||||||
|
t.Fatalf("attempt %d must leave the question armed", i)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
reply, handled := h.resolveClarifyAnswer(ctx, "ну не знаю")
|
||||||
|
if !handled || reply != clarifyGaveUp {
|
||||||
|
t.Fatalf("the fourth try must give up out loud, handled=%v reply=%q", handled, reply)
|
||||||
|
}
|
||||||
|
if reply == "" || strings.Contains(reply, "?") {
|
||||||
|
t.Fatalf("giving up must be spoken and must not be another question: %q", reply)
|
||||||
|
}
|
||||||
|
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||||
|
t.Fatal("a given-up request must leave no armed question")
|
||||||
|
}
|
||||||
|
if reminders, err := st.DueReminders(ctx, h.now().Add(48*time.Hour)); err != nil || len(reminders) != 0 {
|
||||||
|
t.Fatalf("a given-up request must not create anything: reminders=%v err=%v", reminders, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyMaxAttemptsIsConfigurable — one question when the config says one.
|
||||||
|
func TestClarifyMaxAttemptsIsConfigurable(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, _, _ := newClarifyHandler(t)
|
||||||
|
h.clarifyMaxAttempts = 1
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||||
|
t.Fatal("expected a question")
|
||||||
|
}
|
||||||
|
if reply, handled := h.resolveClarifyAnswer(ctx, "ну не знаю"); !handled || reply != clarifyGaveUp {
|
||||||
|
t.Fatalf("with max 1 she must give up at once, handled=%v reply=%q", handled, reply)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyRestatedAnswerWins — «в 11:00», then «нет, в 15:00». The second
|
||||||
|
// value is the one that lands.
|
||||||
|
func TestClarifyRestatedAnswerWins(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни позвонить маме"}, "напомни позвонить маме")); !asked {
|
||||||
|
t.Fatal("expected a question")
|
||||||
|
}
|
||||||
|
// First answer parses, but re-park it by hand as if she had asked again:
|
||||||
|
// what matters here is that Answer prefers the newer value over the parked
|
||||||
|
// one, which is the case the daemon hits on a re-ask.
|
||||||
|
q := h.clarifyStore.Get(voiceDialogueID, h.now())
|
||||||
|
if q == nil {
|
||||||
|
t.Fatal("expected an armed question")
|
||||||
|
}
|
||||||
|
first := h.extractor.Extract(ctx, router.IntentReminder, "в 11:00", h.now())
|
||||||
|
q.Slots = q.Answer("в 11:00", toDialogueSlots(first))
|
||||||
|
|
||||||
|
if reply, handled := h.resolveClarifyAnswer(ctx, "нет, в 15:00"); !handled || reply == clarifyGaveUp {
|
||||||
|
t.Fatalf("the restated answer should complete the request, handled=%v reply=%q", handled, reply)
|
||||||
|
}
|
||||||
|
reminders, err := st.DueReminders(ctx, h.now().Add(48*time.Hour))
|
||||||
|
if err != nil || len(reminders) != 1 {
|
||||||
|
t.Fatalf("expected one reminder: %v err=%v", reminders, err)
|
||||||
|
}
|
||||||
|
want := h.extractor.Extract(ctx, router.IntentReminder, "в 15:00", h.now())
|
||||||
|
if !reminders[0].FireTs.Equal(want.Time) {
|
||||||
|
t.Fatalf("reminder at %v, want the restated %v", reminders[0].FireTs, want.Time)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifiedActOffAllowlistIsStillRefused — clarification fills in an
|
||||||
|
// argument, it never grants authority.
|
||||||
|
func TestClarifiedActOffAllowlistIsStillRefused(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
marker := filepath.Join(t.TempDir(), "not-allowed-ran")
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||||
|
t.Fatal("an act with no fn should be asked about")
|
||||||
|
}
|
||||||
|
reply, handled := h.resolveClarifyAnswer(ctx, "rm "+marker)
|
||||||
|
if !handled {
|
||||||
|
t.Fatal("the answer should be consumed")
|
||||||
|
}
|
||||||
|
if strings.Contains(reply, "готово") {
|
||||||
|
t.Fatalf("an act that is not on the allowlist must not report success: %q", reply)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(marker); !os.IsNotExist(err) {
|
||||||
|
t.Fatalf("a clarified act off the allowlist ran anyway: %v", err)
|
||||||
|
}
|
||||||
|
if tools, err := st.ListTools(ctx, "enabled"); err != nil || len(tools) != 0 {
|
||||||
|
t.Fatalf("clarify must not enable a tool: tools=%+v err=%v", tools, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifiedDestructiveActStillNeedsConfirm — the confirm gate survives the
|
||||||
|
// clarify path.
|
||||||
|
func TestClarifiedDestructiveActStillNeedsConfirm(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, st, _ := newClarifyHandler(t)
|
||||||
|
marker := filepath.Join(t.TempDir(), "destructive-ran")
|
||||||
|
if err := st.EnableTool(ctx, "delete_backups", []string{"touch", marker}, true, "test", h.now()); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentAct, router.Slots{Text: "сделай это"}, "сделай это")); !asked {
|
||||||
|
t.Fatal("expected a question")
|
||||||
|
}
|
||||||
|
reply, handled := h.resolveClarifyAnswer(ctx, "delete_backups")
|
||||||
|
if !handled {
|
||||||
|
t.Fatal("the answer should be consumed")
|
||||||
|
}
|
||||||
|
if !strings.Contains(reply, "да") || h.pending == nil {
|
||||||
|
t.Fatalf("a clarified destructive act must still park a confirm: reply=%q pending=%+v", reply, h.pending)
|
||||||
|
}
|
||||||
|
if _, err := os.Stat(marker); !os.IsNotExist(err) {
|
||||||
|
t.Fatalf("a clarified destructive act ran before confirmation: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestNoQuestionWhenNothingIsMissing — noise keeps the canned reply, so she
|
||||||
|
// never invents a question for nothing.
|
||||||
|
func TestNoQuestionWhenNothingIsMissing(t *testing.T) {
|
||||||
|
h, _, _ := newClarifyHandler(t)
|
||||||
|
for _, dec := range []router.Decision{
|
||||||
|
clarifyDec(router.IntentChat, router.Slots{Text: "эм"}, "эм"),
|
||||||
|
clarifyDec(router.IntentQuery, router.Slots{Text: "ммм"}, "ммм"),
|
||||||
|
clarifyDec(router.IntentNote, router.Slots{Text: "..."}, "..."),
|
||||||
|
} {
|
||||||
|
if question, asked := h.askClarify(dec); asked {
|
||||||
|
t.Fatalf("intent %s should keep the canned reply, got %q", dec.Intent, question)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||||
|
t.Fatal("noise must not park a question")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyExpiryIsAnnouncedAndWordsStillRoute — his answer lands after the
|
||||||
|
// TTL: she must say the old request is gone AND still answer the new words.
|
||||||
|
func TestClarifyExpiryIsAnnouncedAndWordsStillRoute(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
h, _, now := newClarifyHandler(t)
|
||||||
|
emb := router.NewHashEmbedder(1024)
|
||||||
|
h.embedder = emb
|
||||||
|
h.router = buildRouter(emb, h.matcher, 0.55, nil)
|
||||||
|
|
||||||
|
if _, asked := h.askClarify(clarifyDec(router.IntentReminder, router.Slots{Text: "напомни"}, "напомни")); !asked {
|
||||||
|
t.Fatal("expected a question")
|
||||||
|
}
|
||||||
|
*now = now.Add(clarifyTTL + time.Second)
|
||||||
|
|
||||||
|
reply := h.handleText(ctx, "как дела")
|
||||||
|
if !isClarifyExpired(reply) {
|
||||||
|
t.Fatalf("expired question must be announced first, got %q", reply)
|
||||||
|
}
|
||||||
|
if trimClarifyExpired(reply) == "" {
|
||||||
|
t.Fatalf("the new words must still be answered, got only the notice: %q", reply)
|
||||||
|
}
|
||||||
|
if h.clarifyStore.Get(voiceDialogueID, h.now()) != nil {
|
||||||
|
t.Fatal("the expired question must be gone")
|
||||||
|
}
|
||||||
|
// The notice is said once, not on every later utterance.
|
||||||
|
if reply := h.handleText(ctx, "как дела"); isClarifyExpired(reply) {
|
||||||
|
t.Fatalf("notice repeated on a later turn: %q", reply)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestNoPendingQuestionFallsThrough — with nothing parked, an utterance routes
|
||||||
|
// normally.
|
||||||
|
func TestNoPendingQuestionFallsThrough(t *testing.T) {
|
||||||
|
h, _, _ := newClarifyHandler(t)
|
||||||
|
if reply, handled := h.resolveClarifyAnswer(context.Background(), "напомни в 11:00"); handled {
|
||||||
|
t.Fatalf("no open question ⇒ must not be treated as an answer, got %q", reply)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,216 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// pendingHexisExec — a mutating Hexis capability parked awaiting a spoken
|
||||||
|
// confirm. The confirmation is bound to the resolved capability + canonical
|
||||||
|
// target entity so a later "да" can only execute exactly what was proposed
|
||||||
|
// (ecosystem invariant: protected actions require bound confirmation).
|
||||||
|
type pendingHexisExec struct {
|
||||||
|
capabilityID string
|
||||||
|
capName string
|
||||||
|
entityID string
|
||||||
|
displayName string
|
||||||
|
expiry time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
// pendingRoutineConfirm — a proposed routine awaiting a spoken y/n to become
|
||||||
|
// a recurring reminder. Set by detectPattern after creating a proposal.
|
||||||
|
type pendingRoutineConfirm struct {
|
||||||
|
routineID int64
|
||||||
|
action string
|
||||||
|
object string
|
||||||
|
interval float64
|
||||||
|
phrase string
|
||||||
|
expiry time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
// pendingAct — a destructive act awaiting a spoken confirm.
|
||||||
|
type pendingAct struct {
|
||||||
|
fn string
|
||||||
|
args []string
|
||||||
|
phrase string
|
||||||
|
expiry time.Time
|
||||||
|
}
|
||||||
|
|
||||||
|
// confirmTTL — how long a parked destructive confirm stays answerable. Short:
|
||||||
|
// a confirm is a same-breath gesture; a stale prompt shouldn't fire on an
|
||||||
|
// unrelated later "да".
|
||||||
|
const confirmTTL = 90 * time.Second
|
||||||
|
|
||||||
|
// park stores a destructive act awaiting confirmation. Overwrites any prior
|
||||||
|
// pending (last-asked wins — single-user box).
|
||||||
|
func (h *reactiveHandler) park(fn string, args []string, phrase string) {
|
||||||
|
h.mu.Lock()
|
||||||
|
h.pending = &pendingAct{fn: fn, args: args, phrase: phrase, expiry: h.now().Add(confirmTTL)}
|
||||||
|
h.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// resolveConfirm interprets an utterance as the answer to a parked destructive
|
||||||
|
// act OR a parked routine proposal. Returns (reply, true) when it consumed the
|
||||||
|
// utterance as a y/n answer; ("", false) when there's nothing pending (or the
|
||||||
|
// parked act expired), so the caller routes the utterance normally. An
|
||||||
|
// unrecognised answer cancels the pending and routes normally — a confirm that
|
||||||
|
// can't be answered clearly is safer abandoned than left armed.
|
||||||
|
func (h *reactiveHandler) resolveConfirm(ctx context.Context, text string) (string, bool) {
|
||||||
|
h.mu.Lock()
|
||||||
|
defer h.mu.Unlock()
|
||||||
|
|
||||||
|
for _, r := range h.confirmResolvers(ctx) {
|
||||||
|
if !r.claim() {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// The slot is already cleared by claim(): every branch below drops the
|
||||||
|
// pending, including the unclear one — a confirm that can't be
|
||||||
|
// answered clearly is safer abandoned than left armed.
|
||||||
|
switch classifyConfirm(text) {
|
||||||
|
case confirmYes:
|
||||||
|
return r.yes(), true
|
||||||
|
case confirmNo:
|
||||||
|
return r.no(), true
|
||||||
|
default:
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// confirmResolver — one parked-confirm slot in the chain. claim() reports
|
||||||
|
// whether this slot holds a live pending, taking it (and dropping an expired
|
||||||
|
// one) as it goes; yes/no then run the answer. Only ever called with h.mu held.
|
||||||
|
type confirmResolver struct {
|
||||||
|
claim func() bool
|
||||||
|
yes func() string
|
||||||
|
no func() string
|
||||||
|
}
|
||||||
|
|
||||||
|
// confirmResolvers builds the ordered chain resolveConfirm walks. Order is
|
||||||
|
// deliberate: the routine proposal is checked before the tool confirm so a
|
||||||
|
// routine confirm doesn't get eaten by a stale tool pending.
|
||||||
|
func (h *reactiveHandler) confirmResolvers(ctx context.Context) []confirmResolver {
|
||||||
|
var pr *pendingRoutineConfirm
|
||||||
|
var hx *pendingHexisExec
|
||||||
|
var p *pendingAct
|
||||||
|
|
||||||
|
return []confirmResolver{
|
||||||
|
// Routine proposal.
|
||||||
|
{
|
||||||
|
claim: func() bool {
|
||||||
|
pr, h.pendingRoutine = h.pendingRoutine, nil
|
||||||
|
return pr != nil && !h.now().After(pr.expiry)
|
||||||
|
},
|
||||||
|
yes: func() string {
|
||||||
|
// Only record the acceptance. The tick loop reads accepted
|
||||||
|
// routines and nudges on their own interval. Building a
|
||||||
|
// reminder here made a routine fire exactly once (Vikunja #366).
|
||||||
|
if err := h.dataStore.AcceptProposedRoutine(ctx, pr.routineID, h.now()); err != nil {
|
||||||
|
log.Printf("voice: accept proposed routine: %v", err)
|
||||||
|
return "не получилось запомнить рутину."
|
||||||
|
}
|
||||||
|
return "буду напоминать."
|
||||||
|
},
|
||||||
|
no: func() string {
|
||||||
|
if err := h.dataStore.DismissProposedRoutine(ctx, pr.routineID); err != nil {
|
||||||
|
log.Printf("voice: dismiss proposed routine: %v", err)
|
||||||
|
}
|
||||||
|
return "хорошо, не буду."
|
||||||
|
},
|
||||||
|
},
|
||||||
|
// Hexis execution confirm. Bound to the exact capability + target that
|
||||||
|
// was proposed; a stray "да" can only run that, nothing else.
|
||||||
|
{
|
||||||
|
claim: func() bool {
|
||||||
|
hx, h.pendingHexis = h.pendingHexis, nil
|
||||||
|
return hx != nil && !h.now().After(hx.expiry)
|
||||||
|
},
|
||||||
|
yes: func() string {
|
||||||
|
return h.execHexis(ctx, hx.capabilityID, hx.capName, hx.entityID, hx.displayName)
|
||||||
|
},
|
||||||
|
no: func() string { return "отменила." },
|
||||||
|
},
|
||||||
|
// Tool confirm.
|
||||||
|
{
|
||||||
|
claim: func() bool {
|
||||||
|
p, h.pending = h.pending, nil
|
||||||
|
return p != nil && !h.now().After(p.expiry)
|
||||||
|
},
|
||||||
|
yes: func() string {
|
||||||
|
out, err := h.tools.Exec(ctx, p.fn, p.args, true) // confirmed
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: tool %s (confirmed): %v", p.fn, err)
|
||||||
|
if out != "" {
|
||||||
|
return "не получилось выполнить команду: " + firstLine(out)
|
||||||
|
}
|
||||||
|
return "не получилось выполнить команду."
|
||||||
|
}
|
||||||
|
if out != "" {
|
||||||
|
return "готово: " + firstLine(out)
|
||||||
|
}
|
||||||
|
return "готово."
|
||||||
|
},
|
||||||
|
no: func() string { return "отменила." },
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// proposeGap scaffolds a 'proposed' tool for an act whose verb isn't enabled.
|
||||||
|
// maven drafts the registration (name = the verb, provenance = the utterance);
|
||||||
|
// a human enables it on the authed surface. She suggests, never enables.
|
||||||
|
func (h *reactiveHandler) proposeGap(ctx context.Context, dec router.Decision) string {
|
||||||
|
name := firstWord(stripWake(dec.Utterance))
|
||||||
|
if name == "" {
|
||||||
|
return "не разобрала команду — попробуй иначе."
|
||||||
|
}
|
||||||
|
newly, err := h.api.ProposeTool(ctx, name, dec.Utterance, "", h.now())
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: propose tool %q: %v", name, err)
|
||||||
|
return "команды «" + name + "» нет в списке разрешённых."
|
||||||
|
}
|
||||||
|
if newly {
|
||||||
|
return "команды «" + name + "» нет в списке. Предложила её добавить — включи через клиент."
|
||||||
|
}
|
||||||
|
return "команды «" + name + "» пока нет в списке — она уже предложена, включи через клиент."
|
||||||
|
}
|
||||||
|
|
||||||
|
// confirmVerdict — the parse of a y/n confirm answer.
|
||||||
|
type confirmVerdict int
|
||||||
|
|
||||||
|
const (
|
||||||
|
confirmUnknown confirmVerdict = iota
|
||||||
|
confirmYes
|
||||||
|
confirmNo
|
||||||
|
)
|
||||||
|
|
||||||
|
// classifyConfirm reads a short ru/en yes-or-no answer. Substring match on the
|
||||||
|
// stems so inflections/fillers ("да, давай", "нет, отмени") still land.
|
||||||
|
func classifyConfirm(text string) confirmVerdict {
|
||||||
|
t := strings.ToLower(strings.TrimSpace(text))
|
||||||
|
// negatives first — "не надо" contains no "да", but check no-stems before
|
||||||
|
// yes so a leading "нет" isn't shadowed.
|
||||||
|
for _, no := range []string{"нет", "не надо", "отмен", "стоп", "no", "cancel", "stop", "don't"} {
|
||||||
|
if strings.Contains(t, no) {
|
||||||
|
return confirmNo
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, yes := range []string{"да", "ага", "давай", "подтвер", "конечно", "yes", "yeah", "yep", "confirm", "ок", "okay", "ok"} {
|
||||||
|
if strings.Contains(t, yes) {
|
||||||
|
return confirmYes
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return confirmUnknown
|
||||||
|
}
|
||||||
|
|
||||||
|
// actPhrase renders "fn arg1 arg2" for the confirm prompt.
|
||||||
|
func actPhrase(fn string, args []string) string {
|
||||||
|
if len(args) == 0 {
|
||||||
|
return fn
|
||||||
|
}
|
||||||
|
return fn + " " + strings.Join(args, " ")
|
||||||
|
}
|
||||||
@@ -0,0 +1,158 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/loop"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Vikunja #281 — the fourth delivery outcome: a care candidate the restraint
|
||||||
|
// gate suppresses (quiet hours / away / calendar-busy) is not necessarily
|
||||||
|
// lost. If it's worth resurfacing (loop.DigestEligible), it's durably held
|
||||||
|
// (internal/store's digest_entries) and spoken as one bundle once speaking
|
||||||
|
// is appropriate again — never while the suppression reason still holds.
|
||||||
|
|
||||||
|
func breakTrace(blockedBy string) *loop.TickTrace {
|
||||||
|
return &loop.TickTrace{
|
||||||
|
RuleTraces: []loop.RuleTrace{{
|
||||||
|
RuleName: "break",
|
||||||
|
Severity: loop.Sev2,
|
||||||
|
PredicateResult: true,
|
||||||
|
GateResult: false,
|
||||||
|
GateBlockedBy: blockedBy,
|
||||||
|
}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSuppressedCareDigestsAcrossQuietHours — a Sev2 care candidate blocked
|
||||||
|
// by quiet hours is enqueued into the durable digest, and is spoken as a
|
||||||
|
// "digest" nudge only once quiet hours actually end — never while still
|
||||||
|
// suppressed (that would just be a second way to nag through quiet hours).
|
||||||
|
func TestSuppressedCareDigestsAcrossQuietHours(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
|
||||||
|
quiet := loop.State{Now: now, QuietHours: true, Presence: store.Present}
|
||||||
|
tl.enqueueSuppressedDigest(ctx, breakTrace("quiet_hours"), quiet, now)
|
||||||
|
|
||||||
|
entries, err := st.PendingDigestEntries(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pending: %v", err)
|
||||||
|
}
|
||||||
|
if len(entries) != 1 || entries[0].Rule != "break" {
|
||||||
|
t.Fatalf("want 1 pending digest entry for break, got %+v", entries)
|
||||||
|
}
|
||||||
|
|
||||||
|
// still quiet hours: draining now must not speak — the same restraint
|
||||||
|
// that suppressed the live nudge must suppress the bundle too.
|
||||||
|
tl.maybeDrainDigest(ctx, quiet, now)
|
||||||
|
if len(sink.sends) != 0 {
|
||||||
|
t.Fatalf("digest must not drain while quiet hours holds, got %+v", sink.sends)
|
||||||
|
}
|
||||||
|
|
||||||
|
// quiet hours end: this is the moment speaking is appropriate again.
|
||||||
|
after := now.Add(time.Hour)
|
||||||
|
clear := loop.State{Now: after, QuietHours: false, Presence: store.Present}
|
||||||
|
tl.maybeDrainDigest(ctx, clear, after)
|
||||||
|
|
||||||
|
if len(sink.sends) != 1 {
|
||||||
|
t.Fatalf("want exactly 1 dispatched digest bundle, got %d: %+v", len(sink.sends), sink.sends)
|
||||||
|
}
|
||||||
|
if sink.sends[0].RuleName != "digest" {
|
||||||
|
t.Fatalf("want RuleName digest, got %q", sink.sends[0].RuleName)
|
||||||
|
}
|
||||||
|
|
||||||
|
remaining, err := st.PendingDigestEntries(ctx, after)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pending after drain: %v", err)
|
||||||
|
}
|
||||||
|
if len(remaining) != 0 {
|
||||||
|
t.Fatalf("drained entry must no longer be pending, got %+v", remaining)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSuppressedCareDigestDedupesAcrossTicks — quiet hours holding for
|
||||||
|
// several ticks must not enqueue several copies of the same suppressed
|
||||||
|
// nudge; he hears it once when the bundle finally drains.
|
||||||
|
func TestSuppressedCareDigestDedupesAcrossTicks(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
|
||||||
|
quiet := loop.State{Now: now, QuietHours: true, Presence: store.Present}
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
tl.enqueueSuppressedDigest(ctx, breakTrace("quiet_hours"), quiet, now.Add(time.Duration(i)*time.Minute))
|
||||||
|
}
|
||||||
|
|
||||||
|
entries, err := st.PendingDigestEntries(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pending: %v", err)
|
||||||
|
}
|
||||||
|
if len(entries) != 1 {
|
||||||
|
t.Fatalf("3 suppressions of the same nudge must collapse to 1 pending entry, got %d", len(entries))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSuppressedCareDigestExpiresRatherThanDeliveringLate — an entry that
|
||||||
|
// aged out before the suppression cleared is dropped, not spoken late: a
|
||||||
|
// two-day-old "you skipped a break" is noise, not news.
|
||||||
|
func TestSuppressedCareDigestExpiresRatherThanDeliveringLate(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
|
||||||
|
quiet := loop.State{Now: now, QuietHours: true, Presence: store.Present}
|
||||||
|
tl.enqueueSuppressedDigest(ctx, breakTrace("quiet_hours"), quiet, now)
|
||||||
|
|
||||||
|
// well past digestExpiry (24h) before the suppression ever clears.
|
||||||
|
stale := now.Add(48 * time.Hour)
|
||||||
|
tl.expireStaleDigest(ctx, stale)
|
||||||
|
|
||||||
|
clear := loop.State{Now: stale, QuietHours: false, Presence: store.Present}
|
||||||
|
tl.maybeDrainDigest(ctx, clear, stale)
|
||||||
|
|
||||||
|
if len(sink.sends) != 0 {
|
||||||
|
t.Fatalf("a stale digest entry must be dropped, not delivered late; got %+v", sink.sends)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSuppressedCareDigestIgnoresHighSeverity — defense in depth at the
|
||||||
|
// wiring layer: even if a RuleTrace somehow showed a high-severity rule
|
||||||
|
// blocked by a care-only gate reason, the tick driver must not durably
|
||||||
|
// digest it. Alarms bypass the gate and deliver now, unchanged; they must
|
||||||
|
// never be silently delayed into a bundle.
|
||||||
|
func TestSuppressedCareDigestIgnoresHighSeverity(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
|
||||||
|
trace := &loop.TickTrace{RuleTraces: []loop.RuleTrace{{
|
||||||
|
RuleName: "service_down",
|
||||||
|
Severity: loop.Sev4,
|
||||||
|
PredicateResult: true,
|
||||||
|
GateResult: false,
|
||||||
|
GateBlockedBy: "quiet_hours",
|
||||||
|
}}}
|
||||||
|
quiet := loop.State{Now: now, QuietHours: true, Presence: store.Present}
|
||||||
|
tl.enqueueSuppressedDigest(ctx, trace, quiet, now)
|
||||||
|
|
||||||
|
entries, err := st.PendingDigestEntries(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("pending: %v", err)
|
||||||
|
}
|
||||||
|
if len(entries) != 0 {
|
||||||
|
t.Fatalf("high severity must never be digested, got %+v", entries)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,312 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
hexisclient "github.com/kami/hexis/pkg/client"
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// praxisCapability is one arm of the Praxis act dispatch. This is an interface
|
||||||
|
// rather than a map[string]func because each arm carries its own state: the
|
||||||
|
// verb aliases it answers to, the trace name it records, and its own reply
|
||||||
|
// formatting. The dispatch grows an arm per Praxis capability, so a new one is
|
||||||
|
// added to praxisCapabilities below and nothing else changes.
|
||||||
|
type praxisCapability interface {
|
||||||
|
// aliases are the verbs (router fn slots, EN and RU) this capability answers to.
|
||||||
|
aliases() []string
|
||||||
|
// handle runs the capability and returns the user-facing reply.
|
||||||
|
handle(ctx context.Context, h *reactiveHandler, px *praxisClient, dec router.Decision) string
|
||||||
|
}
|
||||||
|
|
||||||
|
// praxisCapabilities is the registry handlePraxisAct consults, in order.
|
||||||
|
var praxisCapabilities = []praxisCapability{
|
||||||
|
listAttentionCapability{},
|
||||||
|
praxisItemAction{
|
||||||
|
verbs: []string{"acknowledge_item", "принято", "понял", "поняла"},
|
||||||
|
ask: "какой пункт отметить принятым?",
|
||||||
|
op: "acknowledge",
|
||||||
|
failure: "не получилось отметить принятым.",
|
||||||
|
success: "принято.",
|
||||||
|
call: func(ctx context.Context, px *praxisClient, id string) error {
|
||||||
|
_, err := px.Acknowledge(ctx, id)
|
||||||
|
return err
|
||||||
|
},
|
||||||
|
},
|
||||||
|
praxisItemAction{
|
||||||
|
verbs: []string{"resolve_item", "сделано", "готово", "решено"},
|
||||||
|
ask: "какой пункт отметить сделанным?",
|
||||||
|
op: "resolve",
|
||||||
|
failure: "не получилось отметить сделанным.",
|
||||||
|
success: "отмечено как сделано.",
|
||||||
|
call: func(ctx context.Context, px *praxisClient, id string) error {
|
||||||
|
_, err := px.Resolve(ctx, id)
|
||||||
|
return err
|
||||||
|
},
|
||||||
|
},
|
||||||
|
praxisItemAction{
|
||||||
|
verbs: []string{"ignore_item", "игнорировать", "неважно"},
|
||||||
|
ask: "какой пункт игнорировать?",
|
||||||
|
op: "ignore",
|
||||||
|
failure: "не получилось проигнорировать.",
|
||||||
|
success: "проигнорировано.",
|
||||||
|
call: func(ctx context.Context, px *praxisClient, id string) error {
|
||||||
|
_, err := px.Ignore(ctx, id)
|
||||||
|
return err
|
||||||
|
},
|
||||||
|
},
|
||||||
|
praxisItemAction{
|
||||||
|
verbs: []string{"pin_item", "закрепить"},
|
||||||
|
ask: "какой пункт закрепить?",
|
||||||
|
op: "pin",
|
||||||
|
failure: "не получилось закрепить.",
|
||||||
|
success: "закреплено.",
|
||||||
|
call: func(ctx context.Context, px *praxisClient, id string) error {
|
||||||
|
_, err := px.Pin(ctx, id, true)
|
||||||
|
return err
|
||||||
|
},
|
||||||
|
},
|
||||||
|
listChangesCapability{},
|
||||||
|
}
|
||||||
|
|
||||||
|
// handlePraxisAct — dispatches ecosystem tool acts through the Praxis tools API.
|
||||||
|
// Returns "" when the act is not a Praxis verb (the caller falls through to the
|
||||||
|
// system command executor). Returns a reply string otherwise.
|
||||||
|
func (h *reactiveHandler) handlePraxisAct(ctx context.Context, dec router.Decision) string {
|
||||||
|
if h.ecosystem == nil || h.ecosystem.praxis == nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
px := h.ecosystem.praxis
|
||||||
|
for _, capability := range praxisCapabilities {
|
||||||
|
for _, alias := range capability.aliases() {
|
||||||
|
if alias == dec.Slots.Fn {
|
||||||
|
return capability.handle(ctx, h, px, dec)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Not a Praxis verb — let the caller fall through.
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// praxisItemAction is the shared shape of the item-lifecycle capabilities: take
|
||||||
|
// an item id from the value slot, call one Praxis endpoint, trace the result.
|
||||||
|
type praxisItemAction struct {
|
||||||
|
verbs []string
|
||||||
|
ask string // reply when no item id was given
|
||||||
|
op string // trace + log name of the operation
|
||||||
|
failure string // reply when the Praxis call errors
|
||||||
|
success string
|
||||||
|
call func(ctx context.Context, px *praxisClient, id string) error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (a praxisItemAction) aliases() []string { return a.verbs }
|
||||||
|
|
||||||
|
func (a praxisItemAction) handle(ctx context.Context, h *reactiveHandler, px *praxisClient, dec router.Decision) string {
|
||||||
|
id := dec.Slots.Value
|
||||||
|
if id == "" {
|
||||||
|
return a.ask
|
||||||
|
}
|
||||||
|
if err := a.call(ctx, px, id); err != nil {
|
||||||
|
log.Printf("ecosystem: praxis %s %s: %v", a.op, id, err)
|
||||||
|
return a.failure
|
||||||
|
}
|
||||||
|
h.recordPraxisTrace(ctx, a.op, map[string]any{"item_id": id})
|
||||||
|
return a.success
|
||||||
|
}
|
||||||
|
|
||||||
|
// listAttentionCapability reads the attention digest and surfaces every item it speaks.
|
||||||
|
type listAttentionCapability struct{}
|
||||||
|
|
||||||
|
func (listAttentionCapability) aliases() []string {
|
||||||
|
return []string{"list_attention", "attention", "внимание", "что требует внимания", "что нового"}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (listAttentionCapability) handle(ctx context.Context, h *reactiveHandler, px *praxisClient, _ router.Decision) string {
|
||||||
|
items, err := px.ListAttention(ctx, 20)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("ecosystem: praxis attention: %v", err)
|
||||||
|
return "не могу сейчас узнать, что требует внимания."
|
||||||
|
}
|
||||||
|
if len(items) == 0 {
|
||||||
|
return "ничего не требует внимания."
|
||||||
|
}
|
||||||
|
h.recordPraxisTrace(ctx, "list_attention", map[string]any{"count": len(items)})
|
||||||
|
var parts []string
|
||||||
|
for _, item := range items {
|
||||||
|
title, _ := item["title"].(string)
|
||||||
|
// importance arrives as JSON number ⇒ float64 over the HTTP contract.
|
||||||
|
importance, _ := item["importance"].(float64)
|
||||||
|
rule, _ := item["rule"].(string)
|
||||||
|
s := title
|
||||||
|
if importance > 0 {
|
||||||
|
s += fmt.Sprintf(" (важность %d", int(importance))
|
||||||
|
if rule != "" {
|
||||||
|
s += ": " + rule
|
||||||
|
}
|
||||||
|
s += ")"
|
||||||
|
}
|
||||||
|
parts = append(parts, s)
|
||||||
|
|
||||||
|
// Speaking an item surfaces it, it does not acknowledge it
|
||||||
|
// (ECOSYSTEM-SPEC.md §2.3: surfaced != acknowledged). Best-effort:
|
||||||
|
// a failed surface call must not block delivering the digest.
|
||||||
|
if id, ok := item["id"].(string); ok && id != "" {
|
||||||
|
if _, err := px.Surface(ctx, id); err != nil {
|
||||||
|
log.Printf("ecosystem: praxis surface %s: %v", id, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "требует внимания: " + strings.Join(parts, "; ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// listChangesCapability reads the recent-changes feed.
|
||||||
|
type listChangesCapability struct{}
|
||||||
|
|
||||||
|
func (listChangesCapability) aliases() []string {
|
||||||
|
return []string{"list_changes", "changes", "изменения", "что изменилось"}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (listChangesCapability) handle(ctx context.Context, h *reactiveHandler, px *praxisClient, _ router.Decision) string {
|
||||||
|
changes, err := px.ListChanges(ctx, 20)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("ecosystem: praxis changes: %v", err)
|
||||||
|
return "не могу сейчас узнать об изменениях."
|
||||||
|
}
|
||||||
|
if len(changes) == 0 {
|
||||||
|
return "нет изменений."
|
||||||
|
}
|
||||||
|
h.recordPraxisTrace(ctx, "list_changes", map[string]any{"count": len(changes)})
|
||||||
|
var parts []string
|
||||||
|
for _, c := range changes {
|
||||||
|
title, _ := c["title"].(string)
|
||||||
|
typ, _ := c["change_type"].(string)
|
||||||
|
parts = append(parts, fmt.Sprintf("%s (%s)", title, typ))
|
||||||
|
}
|
||||||
|
return "изменения: " + strings.Join(parts, "; ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// recordPraxisTrace — writes a fact recording a cross-service ecosystem call.
|
||||||
|
// The fact is stored with source "praxis:trace" so the proactive loop can
|
||||||
|
// reference it and the dashboard can display recent ecosystem activity.
|
||||||
|
func (h *reactiveHandler) recordPraxisTrace(ctx context.Context, operation string, details map[string]any) {
|
||||||
|
now := h.now()
|
||||||
|
value := operation
|
||||||
|
if len(details) > 0 {
|
||||||
|
if b, err := json.Marshal(details); err == nil {
|
||||||
|
value = operation + " " + string(b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_, _ = h.api.WriteFact(ctx, ipc.WriteFactReq{
|
||||||
|
Ts: now,
|
||||||
|
Kind: "system",
|
||||||
|
Key: "praxis:" + operation,
|
||||||
|
Value: value,
|
||||||
|
Source: "praxis:trace",
|
||||||
|
Confidence: 1.0,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleHexisAct — resolves entity references through Nexus and executes
|
||||||
|
// matching capabilities through Hexis. Returns a reply string when handled,
|
||||||
|
// or "" to fall through to the system command executor.
|
||||||
|
func (h *reactiveHandler) handleHexisAct(ctx context.Context, dec router.Decision) string {
|
||||||
|
if h.ecosystem == nil {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// Resolve the utterance text as an entity reference through Nexus. An
|
||||||
|
// ambiguous match must stop and clarify — never guess a mutation target.
|
||||||
|
entityID, displayName, ambiguous, err := h.ecosystem.resolveEntityReference(ctx, dec.Slots.Text, nil)
|
||||||
|
if err != nil {
|
||||||
|
// A genuine Nexus dependency failure, not "no such entity" — stop here
|
||||||
|
// and report degradation rather than silently falling through to the
|
||||||
|
// local command executor (ECOSYSTEM-SPEC.md: services degrade
|
||||||
|
// independently, never a silent all-clear).
|
||||||
|
return "экосистема недоступна, попробуй ещё раз."
|
||||||
|
}
|
||||||
|
if len(ambiguous) > 0 {
|
||||||
|
return "уточни, что именно: " + strings.Join(ambiguous, ", ") + "?"
|
||||||
|
}
|
||||||
|
if entityID == "" {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// Discover Hexis capabilities for this entity. A resolved entity with a
|
||||||
|
// genuine Hexis failure must not be treated as "no capabilities" and
|
||||||
|
// fall through to unrelated local execution.
|
||||||
|
caps, err := h.ecosystem.discoverCapabilities(ctx, entityID)
|
||||||
|
if err != nil {
|
||||||
|
return "экосистема недоступна, попробуй ещё раз."
|
||||||
|
}
|
||||||
|
if len(caps) == 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match the user's verb to a capability by name/description. Collect all
|
||||||
|
// matches: more than one is itself ambiguous, so we ask rather than pick
|
||||||
|
// the first (ecosystem invariant: no arbitrary target for mutation).
|
||||||
|
verb := dec.Slots.Fn
|
||||||
|
if verb == "" {
|
||||||
|
verb = dec.Slots.Text
|
||||||
|
}
|
||||||
|
verbLower := strings.ToLower(verb)
|
||||||
|
|
||||||
|
var matches []*hexisclient.Capability
|
||||||
|
for i, c := range caps {
|
||||||
|
if strings.Contains(strings.ToLower(c.Name), verbLower) ||
|
||||||
|
(c.Description != "" && strings.Contains(strings.ToLower(c.Description), verbLower)) {
|
||||||
|
matches = append(matches, &caps[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(matches) == 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
if len(matches) > 1 {
|
||||||
|
var names []string
|
||||||
|
for _, m := range matches {
|
||||||
|
names = append(names, m.Name)
|
||||||
|
}
|
||||||
|
return "какую команду для " + displayName + ": " + strings.Join(names, ", ") + "?"
|
||||||
|
}
|
||||||
|
matched := matches[0]
|
||||||
|
|
||||||
|
// Read-only capabilities run immediately; mutating ones are parked for an
|
||||||
|
// explicit spoken confirm bound to this capability + target.
|
||||||
|
if !matched.ReadOnly {
|
||||||
|
h.mu.Lock()
|
||||||
|
h.pendingHexis = &pendingHexisExec{
|
||||||
|
capabilityID: matched.ID,
|
||||||
|
capName: matched.Name,
|
||||||
|
entityID: entityID,
|
||||||
|
displayName: displayName,
|
||||||
|
expiry: h.now().Add(confirmTTL),
|
||||||
|
}
|
||||||
|
h.mu.Unlock()
|
||||||
|
return "выполнить «" + matched.Name + "» для " + displayName + "? скажи «да» или «нет»."
|
||||||
|
}
|
||||||
|
|
||||||
|
return h.execHexis(ctx, matched.ID, matched.Name, entityID, displayName)
|
||||||
|
}
|
||||||
|
|
||||||
|
// execHexis runs a resolved capability and records a cross-service trace with
|
||||||
|
// the correlation ID. It reports command success, never operational recovery
|
||||||
|
// (Praxis observes recovery independently).
|
||||||
|
func (h *reactiveHandler) execHexis(ctx context.Context, capID, capName, entityID, displayName string) string {
|
||||||
|
correlationID, err := h.ecosystem.executeCapability(ctx, capID, entityID, nil)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("ecosystem: hexis execute error (cor=%s): %v", correlationID, err)
|
||||||
|
return "не получилось выполнить команду для " + displayName + "."
|
||||||
|
}
|
||||||
|
h.recordPraxisTrace(ctx, "hexis:"+capName, map[string]any{
|
||||||
|
"entity_id": entityID,
|
||||||
|
"entity_name": displayName,
|
||||||
|
"capability": capName,
|
||||||
|
"correlation_id": correlationID,
|
||||||
|
})
|
||||||
|
return "команда выполнена для " + displayName + "."
|
||||||
|
}
|
||||||
@@ -0,0 +1,28 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestPickLLMRouterOff(t *testing.T) {
|
||||||
|
if r := pickLLMRouter(false, llm.New("http://127.0.0.1:1", time.Second)); r != nil {
|
||||||
|
t.Error("flag off should give no LLM router")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The operator can turn the flag on without an LLM phraser configured. That must
|
||||||
|
// leave the classifier running, not panic.
|
||||||
|
func TestPickLLMRouterOnWithoutClient(t *testing.T) {
|
||||||
|
if r := pickLLMRouter(true, nil); r != nil {
|
||||||
|
t.Error("no llama-server should give no LLM router")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPickLLMRouterOn(t *testing.T) {
|
||||||
|
if r := pickLLMRouter(true, llm.New("http://127.0.0.1:1", time.Second)); r == nil {
|
||||||
|
t.Error("flag on with a client should give an LLM router")
|
||||||
|
}
|
||||||
|
}
|
||||||
+91
-124
@@ -57,10 +57,11 @@ import (
|
|||||||
"github.com/kami/maven/internal/delivery/ntfysink"
|
"github.com/kami/maven/internal/delivery/ntfysink"
|
||||||
"github.com/kami/maven/internal/delivery/telegramsink"
|
"github.com/kami/maven/internal/delivery/telegramsink"
|
||||||
"github.com/kami/maven/internal/ipc"
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/loop"
|
||||||
|
"github.com/kami/maven/internal/persona"
|
||||||
"github.com/kami/maven/internal/phraser"
|
"github.com/kami/maven/internal/phraser"
|
||||||
"github.com/kami/maven/internal/store"
|
"github.com/kami/maven/internal/store"
|
||||||
"github.com/kami/maven/internal/webauthn"
|
"github.com/kami/maven/internal/webauthn"
|
||||||
"github.com/kami/maven/internal/loop"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
var errLocked = errors.New("mavend: daemon locked — complete passkey assertion first")
|
var errLocked = errors.New("mavend: daemon locked — complete passkey assertion first")
|
||||||
@@ -96,97 +97,12 @@ func main() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// lockedAPI is a dummy CoreAPI used while the daemon is locked. Every method
|
|
||||||
// returns errLocked. The wire protocol's StoreAPI methods all go through the
|
|
||||||
// Server dispatch on CoreAPI, so returning errLocked from each is correct.
|
|
||||||
type lockedAPI struct{}
|
|
||||||
|
|
||||||
var _ ipc.CoreAPI = (*lockedAPI)(nil)
|
|
||||||
|
|
||||||
func (l *lockedAPI) WriteFact(ctx context.Context, req ipc.WriteFactReq) (int64, error) {
|
|
||||||
return 0, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) LatestFact(ctx context.Context, key string) (ipc.Fact, error) {
|
|
||||||
return ipc.Fact{}, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) LatestFactBySource(ctx context.Context, key, source string) (ipc.Fact, error) {
|
|
||||||
return ipc.Fact{}, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) Since(ctx context.Context, key string, now time.Time) (time.Duration, error) {
|
|
||||||
return 0, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) Presence(ctx context.Context) (ipc.Presence, error) {
|
|
||||||
return ipc.Presence{}, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) CreateReminder(ctx context.Context, fire time.Time, payload, cron string) (int64, error) {
|
|
||||||
return 0, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) MarkReminder(ctx context.Context, id int64, status string) error {
|
|
||||||
return errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) ListReminders(ctx context.Context, n int) ([]ipc.Reminder, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RecordNudge(ctx context.Context, rule, channel, message string, ts time.Time) (int64, error) {
|
|
||||||
return 0, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) ResolveNudge(ctx context.Context, id int64, outcome string, ts time.Time) error {
|
|
||||||
return errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RecentOutcomes(ctx context.Context, rule string, n int) ([]string, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RecentFacts(ctx context.Context, n int) ([]ipc.Fact, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) CalendarEvents(ctx context.Context, from, to time.Time) ([]ipc.Fact, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RecentNudges(ctx context.Context, n int) ([]ipc.Nudge, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) {
|
|
||||||
return 0, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) QueryNotes(ctx context.Context, embedding []float32, k int) ([]ipc.Note, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RecentNotes(ctx context.Context, n int) ([]ipc.Note, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) ProposeTool(ctx context.Context, name, utterance, scope string, ts time.Time) (bool, error) {
|
|
||||||
return false, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) EnableTool(ctx context.Context, name string, cmd []string, destructive bool, scope string, ts time.Time) error {
|
|
||||||
return errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) DisableTool(ctx context.Context, name string) error { return errLocked }
|
|
||||||
func (l *lockedAPI) DeleteTool(ctx context.Context, name string) error { return errLocked }
|
|
||||||
func (l *lockedAPI) ListProposedRoutines(ctx context.Context) ([]ipc.ProposedRoutine, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) DismissProposedRoutine(ctx context.Context, id int64) error { return errLocked }
|
|
||||||
func (l *lockedAPI) LookupTool(ctx context.Context, name string) (ipc.Tool, error) {
|
|
||||||
return ipc.Tool{}, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) ListTools(ctx context.Context, status string) ([]ipc.Tool, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) RevertFact(ctx context.Context, key string) (int64, error) { return 0, errLocked }
|
|
||||||
func (l *lockedAPI) Chat(ctx context.Context, text string) (string, error) {
|
|
||||||
return "", errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) TickTrace(ctx context.Context) (ipc.TickTrace, error) {
|
|
||||||
return ipc.TickTrace{}, errLocked
|
|
||||||
}
|
|
||||||
func (l *lockedAPI) MorningStatus(ctx context.Context) ([]ipc.MorningRoutineStatus, error) {
|
|
||||||
return nil, errLocked
|
|
||||||
}
|
|
||||||
|
|
||||||
func run(args []string) error {
|
func run(args []string) error {
|
||||||
cfgPath := flag.String("config", defaultConfigPath(), "path to mavend JSON config")
|
cfgPath := flag.String("config", defaultConfigPath(), "path to mavend JSON config")
|
||||||
wrappedKeyPath := flag.String("wrapped-key-file", "", "path to wrapped encryption key blob (enables cold-start unlock)")
|
wrappedKeyPath := flag.String("wrapped-key-file", "", "path to wrapped encryption key blob (enables cold-start unlock)")
|
||||||
|
reembed := flag.Bool("reembed", false, "re-embed every stored note and fact with the configured embedder, then serve normally (run once after an embedder swap)")
|
||||||
flag.CommandLine.Parse(args)
|
flag.CommandLine.Parse(args)
|
||||||
|
reembedOnStart = *reembed
|
||||||
cfg, err := config.Load(*cfgPath)
|
cfg, err := config.Load(*cfgPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return err
|
return err
|
||||||
@@ -244,15 +160,16 @@ func run(args []string) error {
|
|||||||
// ----- daemon components (only wired when unlocked) -----
|
// ----- daemon components (only wired when unlocked) -----
|
||||||
// Pre-declare so the unlock path can wire them later.
|
// Pre-declare so the unlock path can wire them later.
|
||||||
var (
|
var (
|
||||||
gatherer *loop.Gatherer
|
gatherer *loop.Gatherer
|
||||||
rules []loop.Rule
|
rules []loop.Rule
|
||||||
phr phraser.Phraser
|
phr phraser.Phraser
|
||||||
voiceW *voiceWiring
|
voiceW *voiceWiring
|
||||||
dispatcher *delivery.Dispatcher
|
dispatcher *delivery.Dispatcher
|
||||||
tl *tickLoop
|
tl *tickLoop
|
||||||
coreAPI ipc.CoreAPI
|
coreAPI ipc.CoreAPI
|
||||||
eco *ecosystemWiring
|
eco *ecosystemWiring
|
||||||
factWorker *factEnrichmentWorker
|
factWorker *factEnrichmentWorker
|
||||||
|
evalWorker *memoryEvalWorker // nil ⇒ memory evaluation off (the default)
|
||||||
)
|
)
|
||||||
|
|
||||||
if !locked {
|
if !locked {
|
||||||
@@ -266,13 +183,14 @@ func run(args []string) error {
|
|||||||
phr = phraser.NewStub()
|
phr = phraser.NewStub()
|
||||||
if cfg.Phraser != nil {
|
if cfg.Phraser != nil {
|
||||||
pc := phraser.Config{
|
pc := phraser.Config{
|
||||||
ModelPath: cfg.Phraser.ModelPath,
|
ModelPath: cfg.Phraser.ModelPath,
|
||||||
BinPath: cfg.Phraser.BinPath,
|
BinPath: cfg.Phraser.BinPath,
|
||||||
Listen: cfg.Phraser.Listen,
|
Listen: cfg.Phraser.Listen,
|
||||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||||
NCtx: cfg.Phraser.NCtx,
|
NCtx: cfg.Phraser.NCtx,
|
||||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||||
Persona: personaFromCfg(cfg),
|
LLMNudges: cfg.Phraser.LLMNudges,
|
||||||
|
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||||
}
|
}
|
||||||
if pc.BinPath == "" {
|
if pc.BinPath == "" {
|
||||||
pc.BinPath = "llama-server"
|
pc.BinPath = "llama-server"
|
||||||
@@ -344,8 +262,9 @@ func run(args []string) error {
|
|||||||
tickInterval := time.Duration(cfg.TickInterval)
|
tickInterval := time.Duration(cfg.TickInterval)
|
||||||
repeatInterval := time.Duration(cfg.RepeatInterval)
|
repeatInterval := time.Duration(cfg.RepeatInterval)
|
||||||
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
||||||
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines))
|
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals)
|
||||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||||
|
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||||
|
|
||||||
coreAPI = &daemonAPI{
|
coreAPI = &daemonAPI{
|
||||||
CoreAPI: ipc.NewStoreAPI(st),
|
CoreAPI: ipc.NewStoreAPI(st),
|
||||||
@@ -357,8 +276,13 @@ func run(args []string) error {
|
|||||||
api.chatFn = voiceW.handler.handleText
|
api.chatFn = voiceW.handler.handleText
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// locked mode: dummy CoreAPI that returns errLocked for everything
|
// locked mode: no real store yet, so there's no meaningful CoreAPI to
|
||||||
coreAPI = &lockedAPI{}
|
// serve. srv.Check below is the actual guard — every CoreAPI call is
|
||||||
|
// refused before it reaches this value. This is just a safe non-nil
|
||||||
|
// placeholder: if the guard is ever bypassed by a bug, calls land
|
||||||
|
// here and fail loudly with ipc.ErrNotImplemented instead of a nil
|
||||||
|
// dereference or, worse, silently succeeding.
|
||||||
|
coreAPI = ipc.UnimplementedCoreAPI{}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- IPC boundary (core ↔ modules) -----
|
// ----- IPC boundary (core ↔ modules) -----
|
||||||
@@ -369,7 +293,17 @@ func run(args []string) error {
|
|||||||
|
|
||||||
passkeySess := webauthn.NewPasskeySession(5 * time.Minute)
|
passkeySess := webauthn.NewPasskeySession(5 * time.Minute)
|
||||||
|
|
||||||
// Set Server.Check — in locked mode, block everything except unlock-path methods.
|
// Set Server.Check — the single authorization guard, run once by
|
||||||
|
// Server.dispatch before any CoreAPI method is called (see
|
||||||
|
// internal/ipc/server.go). In locked mode this is the ONLY thing
|
||||||
|
// standing between an unauthenticated caller and the store: it must
|
||||||
|
// default-deny, with an explicit allowlist for the two methods the
|
||||||
|
// unlock flow itself needs (MethodAssertStepUp, MethodUnlock — neither
|
||||||
|
// of which touches CoreAPI; dispatch handles them directly via
|
||||||
|
// srv.StepUp/srv.UnlockFn). Forgetting to allowlist a new unlock-path
|
||||||
|
// method fails safe (denied); forgetting to guard a new CoreAPI method
|
||||||
|
// is impossible because there is nothing left to forget — every method
|
||||||
|
// not in the allowlist is refused by construction.
|
||||||
if locked {
|
if locked {
|
||||||
srv.Check = func(ctx context.Context, m ipc.Method, _ json.RawMessage) error {
|
srv.Check = func(ctx context.Context, m ipc.Method, _ json.RawMessage) error {
|
||||||
switch m {
|
switch m {
|
||||||
@@ -436,13 +370,14 @@ func run(args []string) error {
|
|||||||
phr = phraser.NewStub()
|
phr = phraser.NewStub()
|
||||||
if cfg.Phraser != nil {
|
if cfg.Phraser != nil {
|
||||||
pc := phraser.Config{
|
pc := phraser.Config{
|
||||||
ModelPath: cfg.Phraser.ModelPath,
|
ModelPath: cfg.Phraser.ModelPath,
|
||||||
BinPath: cfg.Phraser.BinPath,
|
BinPath: cfg.Phraser.BinPath,
|
||||||
Listen: cfg.Phraser.Listen,
|
Listen: cfg.Phraser.Listen,
|
||||||
NGpuLayers: cfg.Phraser.NGpuLayers,
|
NGpuLayers: cfg.Phraser.NGpuLayers,
|
||||||
NCtx: cfg.Phraser.NCtx,
|
NCtx: cfg.Phraser.NCtx,
|
||||||
Timeout: time.Duration(cfg.Phraser.Timeout),
|
Timeout: time.Duration(cfg.Phraser.Timeout),
|
||||||
Persona: personaFromCfg(cfg),
|
LLMNudges: cfg.Phraser.LLMNudges,
|
||||||
|
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||||
}
|
}
|
||||||
if pc.BinPath == "" {
|
if pc.BinPath == "" {
|
||||||
pc.BinPath = "llama-server"
|
pc.BinPath = "llama-server"
|
||||||
@@ -505,10 +440,11 @@ func run(args []string) error {
|
|||||||
tickInterval := time.Duration(cfg.TickInterval)
|
tickInterval := time.Duration(cfg.TickInterval)
|
||||||
repeatInterval := time.Duration(cfg.RepeatInterval)
|
repeatInterval := time.Duration(cfg.RepeatInterval)
|
||||||
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
autotuneInterval := time.Duration(cfg.AutotuneInterval)
|
||||||
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines))
|
tl = newTickLoop(st, gatherer, dispatcher, phr, rules, tickInterval, repeatInterval, autotuneInterval, cfg.Digest, routinesFromConfig(cfg.Routines), config.MorningRoutinesFromConfig(cfg.MorningRoutines), cfg.PatternProposals)
|
||||||
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
factWorker = newFactEnrichmentWorker(st, eco, time.Duration(cfg.FactEnrichmentInterval))
|
||||||
|
evalWorker = newMemoryEvalWorker(st, phr, cfg)
|
||||||
|
|
||||||
// Swap the CoreAPI from lockedAPI to the real store adapter.
|
// Swap the CoreAPI from the locked placeholder to the real store adapter.
|
||||||
newAPI := &daemonAPI{
|
newAPI := &daemonAPI{
|
||||||
CoreAPI: ipc.NewStoreAPI(st),
|
CoreAPI: ipc.NewStoreAPI(st),
|
||||||
getTrace: tl.trace,
|
getTrace: tl.trace,
|
||||||
@@ -543,6 +479,13 @@ func run(args []string) error {
|
|||||||
factWorker.run(ctx)
|
factWorker.run(ctx)
|
||||||
}()
|
}()
|
||||||
|
|
||||||
|
// Start background memory evaluation (nil unless configured).
|
||||||
|
if evalWorker != nil {
|
||||||
|
go func() {
|
||||||
|
evalWorker.run(ctx)
|
||||||
|
}()
|
||||||
|
}
|
||||||
|
|
||||||
dl.unlock()
|
dl.unlock()
|
||||||
log.Printf("mavend: unlocked via passkey assertion")
|
log.Printf("mavend: unlocked via passkey assertion")
|
||||||
return nil
|
return nil
|
||||||
@@ -581,6 +524,13 @@ func run(args []string) error {
|
|||||||
defer wg.Done()
|
defer wg.Done()
|
||||||
factWorker.run(ctx)
|
factWorker.run(ctx)
|
||||||
}()
|
}()
|
||||||
|
if evalWorker != nil {
|
||||||
|
wg.Add(1)
|
||||||
|
go func() {
|
||||||
|
defer wg.Done()
|
||||||
|
evalWorker.run(ctx)
|
||||||
|
}()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
<-ctx.Done()
|
<-ctx.Done()
|
||||||
@@ -596,12 +546,29 @@ func run(args []string) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// personaFromCfg extracts the voice persona from the config, or returns ""
|
// personaFacts reads the optional, deployment-specific facts (his name, his
|
||||||
// when voice isn't configured. Used to pass a character prompt into the
|
// city, the free-text persona string) out of the config. Everything here may
|
||||||
// LLM phraser without requiring voice to be enabled.
|
// be empty — the context block is correct without any of it.
|
||||||
func personaFromCfg(cfg *config.Config) string {
|
func personaFacts(cfg *config.Config) persona.Facts {
|
||||||
if cfg.Voice != nil {
|
f := persona.Facts{
|
||||||
return cfg.Voice.Persona
|
// Telegram lives outside the voice block, so it counts either way.
|
||||||
|
Telegram: cfg.Telegram != nil && cfg.Telegram.BotToken != "" && cfg.Telegram.ChatID != "",
|
||||||
}
|
}
|
||||||
return ""
|
if cfg.Voice == nil {
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
f.OwnerName = cfg.Voice.OwnerName
|
||||||
|
f.City = cfg.Voice.City
|
||||||
|
f.Static = cfg.Voice.Persona
|
||||||
|
// Same test wireVoice uses to pick the real provider over the stub.
|
||||||
|
f.Weather = cfg.Voice.Weather != nil && cfg.Voice.Weather.Provider == "open-meteo"
|
||||||
|
f.Tools = len(cfg.Voice.Tools) > 0
|
||||||
|
return f
|
||||||
|
}
|
||||||
|
|
||||||
|
// contextBlockFn returns the per-turn renderer of the shared context block.
|
||||||
|
// Per turn, not once at startup, because the block states the current time.
|
||||||
|
func contextBlockFn(cfg *config.Config, now func() time.Time) func() string {
|
||||||
|
f := personaFacts(cfg)
|
||||||
|
return func() string { return f.Block(now()) }
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,88 @@
|
|||||||
|
// mavend/memoryeval.go — the driver for background memory evaluation
|
||||||
|
// (Vikunja #248). The evaluator itself is pure-ish and lives in
|
||||||
|
// internal/memeval; this is the one impure part: a ticker, the store, and the
|
||||||
|
// resident model's base URL.
|
||||||
|
//
|
||||||
|
// It is its own goroutine and NOT a step on the main tick, deliberately. The
|
||||||
|
// tick runs every 60s and has a delivery deadline behind it; an evaluation is
|
||||||
|
// a multi-second LLM round-trip on the same llama-server that answers voice
|
||||||
|
// turns, and it happens hourly at most. Bolting it onto the tick would make
|
||||||
|
// every hour's tick the slow one for no benefit.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/config"
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
"github.com/kami/maven/internal/memeval"
|
||||||
|
"github.com/kami/maven/internal/phraser"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// memoryEvalWorker — ticker + evaluator.
|
||||||
|
type memoryEvalWorker struct {
|
||||||
|
eval *memeval.Evaluator
|
||||||
|
interval time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
// newMemoryEvalWorker wires the evaluation loop, or returns nil when it should
|
||||||
|
// not run at all. nil is the normal case and every caller must handle it:
|
||||||
|
//
|
||||||
|
// - no memory_eval config block ⇒ off (a capability is off unless configured);
|
||||||
|
// - no LLM phraser ⇒ nothing to evaluate with. There is no template fallback
|
||||||
|
// here on purpose: a "memory evaluation" assembled from string templates
|
||||||
|
// would be a fixed sentence pretending to be an observation.
|
||||||
|
func newMemoryEvalWorker(st *store.Store, phr phraser.Phraser, cfg *config.Config) *memoryEvalWorker {
|
||||||
|
if cfg.MemoryEval == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
lp, ok := phr.(*phraser.LLMPhraser)
|
||||||
|
if !ok {
|
||||||
|
log.Printf("memory eval: configured but no llama-server phraser — evaluation disabled")
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
interval := time.Duration(cfg.MemoryEval.Interval)
|
||||||
|
if interval <= 0 {
|
||||||
|
interval = config.DefaultMemoryEvalInterval
|
||||||
|
}
|
||||||
|
// A generous per-request timeout: this is a long prompt to a Thinking model
|
||||||
|
// and nobody is waiting on the answer.
|
||||||
|
client := llm.New(lp.BaseURL(), 5*time.Minute)
|
||||||
|
ev := memeval.NewEvaluator(st, st, client, memeval.Config{
|
||||||
|
MaxItems: cfg.MemoryEval.MaxItems,
|
||||||
|
MinConfidence: cfg.MemoryEval.MinConfidence,
|
||||||
|
ContextBlock: contextBlockFn(cfg, time.Now),
|
||||||
|
})
|
||||||
|
log.Printf("memory eval: enabled, every %s", interval)
|
||||||
|
return &memoryEvalWorker{eval: ev, interval: interval}
|
||||||
|
}
|
||||||
|
|
||||||
|
// run evaluates every interval until ctx is canceled.
|
||||||
|
//
|
||||||
|
// The first evaluation waits a full interval rather than firing at startup, the
|
||||||
|
// opposite of the tick loop's cold-start behaviour. A tick that fires late is a
|
||||||
|
// nudge that arrives late; an evaluation that fires late is nothing at all, and
|
||||||
|
// the alternative is a heavy LLM call competing with startup — including with
|
||||||
|
// the first voice turn after a restart.
|
||||||
|
func (w *memoryEvalWorker) run(ctx context.Context) {
|
||||||
|
ticker := time.NewTicker(w.interval)
|
||||||
|
defer ticker.Stop()
|
||||||
|
for {
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return
|
||||||
|
case now := <-ticker.C:
|
||||||
|
obs, err := w.eval.Evaluate(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("memory eval: %v", err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, o := range obs {
|
||||||
|
log.Printf("memory eval: noted (%.2f, %s): %s", o.Conf, o.Action, o.Text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
// mavend/patterns.go — the shared detect+propose step of pattern inference
|
||||||
|
// (Vikunja #43). Event *extraction* (fact -> action/object) happens at fact-
|
||||||
|
// write time in detectPattern below, tied to whichever channel wrote the
|
||||||
|
// fact. Detection — turning a run of events into a proposed routine — is
|
||||||
|
// channel-agnostic: it only needs what's already in the events table, so it
|
||||||
|
// runs both right after a voice fact-write (for the immediate "напоминать?"
|
||||||
|
// confirmation) and, proactively, from the digestion tick (tick.go's
|
||||||
|
// detectPatterns) over every action+object pair on record, not just the one
|
||||||
|
// that was just talked about.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/pattern"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// detectAndPropose runs the pattern detector over every recorded event for
|
||||||
|
// action+object and, if a stable pattern is found and nothing has been
|
||||||
|
// proposed/accepted/dismissed for this pair yet, creates a proposed_routines
|
||||||
|
// row. Returns (nil, 0, nil) — not an error — whenever there is nothing new
|
||||||
|
// to report: too few events, irregular intervals, or a pair that already has
|
||||||
|
// a row in any status. That last case is the one that matters most: it is
|
||||||
|
// how a routine the owner already DISMISSED stays dismissed forever, because
|
||||||
|
// the row survives dismissal (status flips in place, see
|
||||||
|
// store.DismissProposedRoutine) and both the Lookup check here and the
|
||||||
|
// table's UNIQUE(action, object) constraint refuse to create a second one.
|
||||||
|
func detectAndPropose(ctx context.Context, ds *store.Store, action, object string, ts time.Time) (*pattern.ProposedRoutine, int64, error) {
|
||||||
|
events, err := ds.EventsFor(ctx, action, object)
|
||||||
|
if err != nil {
|
||||||
|
return nil, 0, fmt.Errorf("events for %s/%s: %w", action, object, err)
|
||||||
|
}
|
||||||
|
patEvents := make([]pattern.Event, len(events))
|
||||||
|
for i, e := range events {
|
||||||
|
patEvents[i] = pattern.Event{
|
||||||
|
FactID: e.FactID,
|
||||||
|
Action: e.Action,
|
||||||
|
Object: e.Object,
|
||||||
|
Ts: e.Ts,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
r, err := pattern.Detect(patEvents)
|
||||||
|
if err != nil {
|
||||||
|
return nil, 0, fmt.Errorf("detect %s/%s: %w", action, object, err)
|
||||||
|
}
|
||||||
|
if r == nil {
|
||||||
|
return nil, 0, nil // not enough data or intervals too irregular
|
||||||
|
}
|
||||||
|
|
||||||
|
// Belt: check first so the common "nothing new" case never even attempts
|
||||||
|
// an insert. Suspenders: CreateProposedRoutine's ON CONFLICT DO NOTHING
|
||||||
|
// (backed by the UNIQUE(action,object) constraint) is the actual
|
||||||
|
// guarantee — this Lookup is an optimization, not the source of truth.
|
||||||
|
existing, err := ds.LookupProposedRoutine(ctx, r.Action, r.Object)
|
||||||
|
if err != nil {
|
||||||
|
return nil, 0, fmt.Errorf("lookup proposed routine %s/%s: %w", action, object, err)
|
||||||
|
}
|
||||||
|
if existing != nil {
|
||||||
|
return nil, 0, nil // already proposed, accepted, or dismissed — say nothing
|
||||||
|
}
|
||||||
|
|
||||||
|
id, err := ds.CreateProposedRoutine(ctx, r.Action, r.Object, r.IntervalDays, ts)
|
||||||
|
if err != nil {
|
||||||
|
if errors.Is(err, store.ErrProposedRoutineExists) {
|
||||||
|
return nil, 0, nil // lost a race with another caller — not an error
|
||||||
|
}
|
||||||
|
return nil, 0, fmt.Errorf("create proposed routine %s/%s: %w", action, object, err)
|
||||||
|
}
|
||||||
|
return r, id, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// detectPattern extracts an event from the written fact and runs the pattern
|
||||||
|
// detector. If a stable recurring pattern is found and no proposed routine
|
||||||
|
// exists for this action+object yet, one is created and the user is prompted
|
||||||
|
// to confirm via the park() mechanism. Returns the suggestion phrase when a
|
||||||
|
// new proposal was created and parked; "" otherwise.
|
||||||
|
func (h *reactiveHandler) detectPattern(ctx context.Context, factID int64, key, value string, ts time.Time) string {
|
||||||
|
ev := pattern.Extract(factID, key, value, ts)
|
||||||
|
if ev == nil {
|
||||||
|
return "" // not an actionable event
|
||||||
|
}
|
||||||
|
if _, err := h.dataStore.CreateEvent(ctx, factID, ev.Action, ev.Object, ts); err != nil {
|
||||||
|
log.Printf("voice: create event: %v", err)
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
// Detect+propose (Vikunja #43) is shared with the digestion tick's
|
||||||
|
// proactive scan — see detectAndPropose above. Event *extraction* stays
|
||||||
|
// here, tied to this fact write; detection over the accumulated history does
|
||||||
|
// not need to happen right now for the voice path to have already done
|
||||||
|
// its job — it's dedupe-safe to also let the next tick find the same
|
||||||
|
// pattern independently.
|
||||||
|
r, id, err := detectAndPropose(ctx, h.dataStore, ev.Action, ev.Object, ts)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: detect pattern %s/%s: %v", ev.Action, ev.Object, err)
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
if r == nil {
|
||||||
|
return "" // not enough data, too irregular, or already proposed/decided
|
||||||
|
}
|
||||||
|
log.Printf("voice: proposed routine: %s/%s every %.1f days", r.Action, r.Object, r.IntervalDays)
|
||||||
|
|
||||||
|
// Park the proposal for voice confirmation.
|
||||||
|
phrase := pattern.PhraseRoutine(r)
|
||||||
|
h.mu.Lock()
|
||||||
|
h.pendingRoutine = &pendingRoutineConfirm{
|
||||||
|
routineID: id,
|
||||||
|
action: r.Action,
|
||||||
|
object: r.Object,
|
||||||
|
interval: r.IntervalDays,
|
||||||
|
phrase: phrase,
|
||||||
|
expiry: ts.Add(confirmTTL),
|
||||||
|
}
|
||||||
|
h.mu.Unlock()
|
||||||
|
return phrase
|
||||||
|
}
|
||||||
@@ -0,0 +1,285 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/config"
|
||||||
|
"github.com/kami/maven/internal/delivery"
|
||||||
|
"github.com/kami/maven/internal/loop"
|
||||||
|
"github.com/kami/maven/internal/pattern"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// seedRefillEvents writes N weekly "refill/cat_water" events straight to the
|
||||||
|
// events table — this is what the tick reads, independent of any utterance.
|
||||||
|
func seedRefillEvents(t *testing.T, st *store.Store, ctx context.Context, base time.Time, n int) {
|
||||||
|
t.Helper()
|
||||||
|
for i := 0; i < n; i++ {
|
||||||
|
factID, err := st.WriteFact(ctx, base.Add(time.Duration(i)*7*24*time.Hour), store.KindSelf,
|
||||||
|
"cat_water", "refill", "test", 1.0, sql.NullInt64{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("write fact %d: %v", i, err)
|
||||||
|
}
|
||||||
|
if _, err := st.CreateEvent(ctx, factID, "refill", "cat_water", base.Add(time.Duration(i)*7*24*time.Hour)); err != nil {
|
||||||
|
t.Fatalf("create event %d: %v", i, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickDetectsPatternFromStoredEvents proves the tick notices a pattern on
|
||||||
|
// its own, reading straight from the store — not as a side effect of a live
|
||||||
|
// utterance (Vikunja #43). MinEvents weekly events with no voice turn in
|
||||||
|
// sight must produce exactly one proposed routine.
|
||||||
|
func TestTickDetectsPatternFromStoredEvents(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
|
||||||
|
tl := newTestTickLoop(t, st, &fakeSink{}, nil)
|
||||||
|
tl.detectPatterns(ctx, now, loop.State{})
|
||||||
|
|
||||||
|
rows, err := st.ListProposedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed routines: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 1 {
|
||||||
|
t.Fatalf("proposed routines = %d, want 1: %+v", len(rows), rows)
|
||||||
|
}
|
||||||
|
if rows[0].Action != "refill" || rows[0].Object != "cat_water" {
|
||||||
|
t.Errorf("proposed routine = %s/%s, want refill/cat_water", rows[0].Action, rows[0].Object)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickPatternDetectionIsIdempotent proves running the tick's pattern scan
|
||||||
|
// twice does not spam a second proposal for the same pair, and that the store
|
||||||
|
// itself is what stops the duplicate (not tick-local state) — the whole point
|
||||||
|
// of the guard, since the tick has no memory of what it proposed last time.
|
||||||
|
func TestTickPatternDetectionIsIdempotent(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
|
||||||
|
tl := newTestTickLoop(t, st, &fakeSink{}, nil)
|
||||||
|
tl.detectPatterns(ctx, now, loop.State{})
|
||||||
|
tl.detectPatterns(ctx, now.Add(time.Hour), loop.State{})
|
||||||
|
|
||||||
|
rows, err := st.ListProposedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed routines: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 1 {
|
||||||
|
t.Fatalf("proposed routines after two ticks = %d, want 1 (no duplicate): %+v", len(rows), rows)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickPatternDetectionRespectsDismissal proves the single worst failure
|
||||||
|
// mode here — a proposal the owner already said no to coming back on the next
|
||||||
|
// tick — cannot happen. Dismissal flips the row's status in place; it must
|
||||||
|
// still be there to block re-proposal.
|
||||||
|
func TestTickPatternDetectionRespectsDismissal(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
|
||||||
|
tl := newTestTickLoop(t, st, &fakeSink{}, nil)
|
||||||
|
tl.detectPatterns(ctx, now, loop.State{})
|
||||||
|
|
||||||
|
rows, err := st.ListProposedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed routines: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 1 {
|
||||||
|
t.Fatalf("setup: proposed routines = %d, want 1", len(rows))
|
||||||
|
}
|
||||||
|
if err := st.DismissProposedRoutine(ctx, rows[0].ID); err != nil {
|
||||||
|
t.Fatalf("dismiss: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// More events for the same pair arrive, and the tick runs again — a
|
||||||
|
// dismissed pattern must not resurface.
|
||||||
|
seedRefillEvents(t, st, ctx, now.Add(30*24*time.Hour), pattern.MinEvents)
|
||||||
|
tl.detectPatterns(ctx, now.Add(60*24*time.Hour), loop.State{})
|
||||||
|
|
||||||
|
proposed, err := st.ListProposedRoutinesByStatus(ctx, store.RoutineProposed)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed: %v", err)
|
||||||
|
}
|
||||||
|
if len(proposed) != 0 {
|
||||||
|
t.Fatalf("a dismissed pattern came back: %+v", proposed)
|
||||||
|
}
|
||||||
|
all, err := st.ListProposedRoutinesByStatus(ctx, "")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list all: %v", err)
|
||||||
|
}
|
||||||
|
if len(all) != 1 {
|
||||||
|
t.Fatalf("total rows for the pair = %d, want 1 (still dismissed, not duplicated): %+v", len(all), all)
|
||||||
|
}
|
||||||
|
if all[0].Status != store.RoutineDismissed {
|
||||||
|
t.Errorf("status = %s, want dismissed", all[0].Status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// proposalRule — the rule name announceProposal uses for the seeded pair.
|
||||||
|
const proposalRule = "proposal:refill cat_water"
|
||||||
|
|
||||||
|
// TestTickProposalSilentByDefault — detection is always on, announcing is not.
|
||||||
|
// With no pattern_proposals block the tick still records the proposal, and says
|
||||||
|
// nothing about it: Maven is not autonomous, so a behaviour that speaks without
|
||||||
|
// being asked stays off until it is configured.
|
||||||
|
func TestTickProposalSilentByDefault(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
markPresent(t, st, ctx, now)
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
tl.tick(ctx, now)
|
||||||
|
|
||||||
|
if n := countSends(sink, proposalRule); n != 0 {
|
||||||
|
t.Fatalf("announced %d proposals with no config, want 0", n)
|
||||||
|
}
|
||||||
|
rows, err := st.ListProposedRoutinesByStatus(ctx, store.RoutineProposed)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 1 {
|
||||||
|
t.Fatalf("proposed routines = %d, want 1 (silent, but recorded)", len(rows))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickAnnouncesProposalWhenConfigured — with notify on, the proposal goes
|
||||||
|
// out once through the ordinary delivery path, worded by the detector itself.
|
||||||
|
// Later ticks stay quiet because the pair is already proposed: one pattern is
|
||||||
|
// one announcement, ever.
|
||||||
|
func TestTickAnnouncesProposalWhenConfigured(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
markPresent(t, st, ctx, now)
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
tl.proposalCfg = &config.PatternProposalConfig{Notify: true}
|
||||||
|
tl.tick(ctx, now)
|
||||||
|
|
||||||
|
var got *delivery.Sendable
|
||||||
|
for i := range sink.sends {
|
||||||
|
if sink.sends[i].RuleName == proposalRule {
|
||||||
|
got = &sink.sends[i]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if got == nil {
|
||||||
|
t.Fatalf("proposal was not announced; sends=%+v", sink.sends)
|
||||||
|
}
|
||||||
|
if !strings.Contains(got.Body, "напоминать?") {
|
||||||
|
t.Errorf("body = %q, want the detector's own question", got.Body)
|
||||||
|
}
|
||||||
|
if got.Channel != delivery.ChannelVoice {
|
||||||
|
t.Errorf("channel = %v, want voice (sev1, present)", got.Channel)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A month of further ticks: the pair already has a row, so there is
|
||||||
|
// nothing new to detect and nothing more to say.
|
||||||
|
sink.sends = nil
|
||||||
|
later := now.Add(40 * 24 * time.Hour)
|
||||||
|
markPresent(t, st, ctx, later)
|
||||||
|
tl.tick(ctx, later)
|
||||||
|
if n := countSends(sink, proposalRule); n != 0 {
|
||||||
|
t.Fatalf("re-announced an existing proposal %d times, want 0", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickProposalRespectsGate — a proposal is the least urgent thing Maven can
|
||||||
|
// say, so it is sev1 and the restraint gate suppresses it. Away presence means
|
||||||
|
// it is not announced at all: it is not held, not retried, it just lives on
|
||||||
|
// /routines. The proposal row is still written — noticing is never gated.
|
||||||
|
func TestTickProposalRespectsGate(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
// no presence probes ⇒ away ⇒ care-class gate blocks.
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
tl.proposalCfg = &config.PatternProposalConfig{Notify: true}
|
||||||
|
tl.tick(ctx, now)
|
||||||
|
|
||||||
|
if n := countSends(sink, proposalRule); n != 0 {
|
||||||
|
t.Fatalf("away: announced %d proposals, want 0", n)
|
||||||
|
}
|
||||||
|
if !tl.lastProposalAt.IsZero() {
|
||||||
|
t.Error("cooldown clock advanced on a suppressed announcement")
|
||||||
|
}
|
||||||
|
rows, err := st.ListProposedRoutinesByStatus(ctx, store.RoutineProposed)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 1 {
|
||||||
|
t.Fatalf("proposed routines = %d, want 1 (detection is never gated)", len(rows))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickProposalCooldownSpacesAnnouncements — two patterns detected on the
|
||||||
|
// same tick must not become two interruptions. The second one waits for the
|
||||||
|
// cooldown, and is on /routines meanwhile.
|
||||||
|
func TestTickProposalCooldownSpacesAnnouncements(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedRefillEvents(t, st, ctx, now, pattern.MinEvents)
|
||||||
|
for i := 0; i < pattern.MinEvents; i++ {
|
||||||
|
ts := now.Add(time.Duration(i) * 3 * 24 * time.Hour)
|
||||||
|
factID, err := st.WriteFact(ctx, ts, store.KindSelf, "litter_box", "clean", "test", 1.0, sql.NullInt64{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("write fact: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := st.CreateEvent(ctx, factID, "clean", "litter_box", ts); err != nil {
|
||||||
|
t.Fatalf("create event: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
markPresent(t, st, ctx, now)
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
tl.proposalCfg = &config.PatternProposalConfig{Notify: true, Cooldown: config.Duration(24 * time.Hour)}
|
||||||
|
tl.tick(ctx, now)
|
||||||
|
|
||||||
|
announced := 0
|
||||||
|
for _, s := range sink.sends {
|
||||||
|
if strings.HasPrefix(s.RuleName, "proposal:") {
|
||||||
|
announced++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if announced != 1 {
|
||||||
|
t.Fatalf("announced %d proposals on one tick, want exactly 1", announced)
|
||||||
|
}
|
||||||
|
rows, err := st.ListProposedRoutinesByStatus(ctx, store.RoutineProposed)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("list proposed: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 2 {
|
||||||
|
t.Fatalf("proposed routines = %d, want 2 (both recorded, one announced)", len(rows))
|
||||||
|
}
|
||||||
|
|
||||||
|
// Still inside the cooldown: silence, even though a proposal is pending.
|
||||||
|
sink.sends = nil
|
||||||
|
soon := now.Add(time.Hour)
|
||||||
|
markPresent(t, st, ctx, soon)
|
||||||
|
tl.tick(ctx, soon)
|
||||||
|
for _, s := range sink.sends {
|
||||||
|
if strings.HasPrefix(s.RuleName, "proposal:") {
|
||||||
|
t.Fatalf("announced %q inside the cooldown", s.RuleName)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,158 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"math"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/memory"
|
||||||
|
"github.com/kami/maven/internal/phraser"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
"github.com/kami/maven/internal/voice"
|
||||||
|
)
|
||||||
|
|
||||||
|
// fixedEmbedder hands back a vector chosen per text, so a test can say exactly
|
||||||
|
// how close each stored memory is to the question. The real embedders make
|
||||||
|
// scores that are realistic but not controllable, and this test is about the
|
||||||
|
// gate, not about the embedder.
|
||||||
|
type fixedEmbedder struct{ vecs map[string][]float32 }
|
||||||
|
|
||||||
|
func (f *fixedEmbedder) Dim() int { return 4 }
|
||||||
|
func (f *fixedEmbedder) Close() error { return nil }
|
||||||
|
|
||||||
|
func (f *fixedEmbedder) Embed(_ context.Context, text string) ([]float32, error) {
|
||||||
|
v, ok := f.vecs[text]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("fixedEmbedder: no vector for %q", text)
|
||||||
|
}
|
||||||
|
return v, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// scoreVec builds a unit vector whose cosine against the query vector
|
||||||
|
// (1,0,0,0) is exactly score.
|
||||||
|
func scoreVec(score float64) []float32 {
|
||||||
|
rest := math.Sqrt(1 - score*score)
|
||||||
|
return []float32{float32(score), float32(rest), 0, 0}
|
||||||
|
}
|
||||||
|
|
||||||
|
// recordingPhraser remembers what the query path handed it to phrase, which is
|
||||||
|
// how the test can tell which pass produced the answer.
|
||||||
|
type recordingPhraser struct {
|
||||||
|
*phraser.Stub
|
||||||
|
notes []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (r *recordingPhraser) PhraseQuery(ctx context.Context, utterance string, notes []string) (string, error) {
|
||||||
|
r.notes = notes
|
||||||
|
return r.Stub.PhraseQuery(ctx, utterance, notes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// recallCase — one stored memory: its text, how close it is to the question,
|
||||||
|
// whether it is a note or a fact, and whether the notes table holds it too.
|
||||||
|
type recallCase struct {
|
||||||
|
text string
|
||||||
|
score float64
|
||||||
|
kind string
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildRecallHandler stores the given memories and returns a handler whose
|
||||||
|
// query path can be run directly. Notes go into BOTH the notes table and the
|
||||||
|
// vector index, which is what the daemon does (voice.go's IntentNote).
|
||||||
|
func buildRecallHandler(t *testing.T, question string, mems []recallCase) (*reactiveHandler, *recordingPhraser) {
|
||||||
|
t.Helper()
|
||||||
|
ctx := context.Background()
|
||||||
|
st := newTestStore(t)
|
||||||
|
emb := &fixedEmbedder{vecs: map[string][]float32{question: {1, 0, 0, 0}}}
|
||||||
|
mem := memory.NewInMemoryStore()
|
||||||
|
now := time.Now()
|
||||||
|
|
||||||
|
for i, m := range mems {
|
||||||
|
vec := scoreVec(m.score)
|
||||||
|
emb.vecs[m.text] = vec
|
||||||
|
id := fmt.Sprintf("%s:%d", m.kind, i)
|
||||||
|
if m.kind == "note" {
|
||||||
|
if _, err := st.WriteNote(ctx, now, m.text, vec, "tap:voice"); err != nil {
|
||||||
|
t.Fatalf("WriteNote: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := mem.Insert(ctx, id, vec, map[string]string{"text": m.text, "type": m.kind}); err != nil {
|
||||||
|
t.Fatalf("memory insert: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
phr := &recordingPhraser{Stub: phraser.NewStub()}
|
||||||
|
h := &reactiveHandler{
|
||||||
|
api: ipc.NewStoreAPI(st),
|
||||||
|
embedder: emb,
|
||||||
|
replier: voice.NewStubReplier(),
|
||||||
|
phraser: phr,
|
||||||
|
now: func() time.Time { return now },
|
||||||
|
memStore: mem,
|
||||||
|
dataStore: st,
|
||||||
|
queryMinScore: 0.55,
|
||||||
|
queryMinMargin: 0.008,
|
||||||
|
weatherProvider: nil,
|
||||||
|
}
|
||||||
|
return h, phr
|
||||||
|
}
|
||||||
|
|
||||||
|
func askQuery(t *testing.T, h *reactiveHandler, question string) string {
|
||||||
|
t.Helper()
|
||||||
|
return h.applyAction(context.Background(), router.Decision{
|
||||||
|
Intent: router.IntentQuery,
|
||||||
|
Utterance: question,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestQueryRecallNoteCanWin — the note-recall regression (Vikunja #373). Notes
|
||||||
|
// and facts share one vector index, and a note that clearly beats everything
|
||||||
|
// else must be the answer. Before the fix the memory pass only ran after the
|
||||||
|
// notes-only gate had already rejected the same note at the same score, so only
|
||||||
|
// a fact could ever come back from it.
|
||||||
|
func TestQueryRecallNoteCanWin(t *testing.T) {
|
||||||
|
const q = "где молоко"
|
||||||
|
|
||||||
|
t.Run("a clearly best note answers", func(t *testing.T) {
|
||||||
|
h, phr := buildRecallHandler(t, q, []recallCase{
|
||||||
|
{text: "молоко стоит в холодильнике", score: 0.90, kind: "note"},
|
||||||
|
{text: "выучил пару аккордов", score: 0.50, kind: "note"},
|
||||||
|
})
|
||||||
|
reply := askQuery(t, h, q)
|
||||||
|
if want := "вот что я нашла: молоко стоит в холодильнике"; reply != want {
|
||||||
|
t.Errorf("reply %q, want %q", reply, want)
|
||||||
|
}
|
||||||
|
// One text, the winning memory's — the answer came from the memory
|
||||||
|
// pass, not from handing the phraser every note in the table.
|
||||||
|
if len(phr.notes) != 1 || phr.notes[0] != "молоко стоит в холодильнике" {
|
||||||
|
t.Errorf("phraser got %q, want just the recalled note", phr.notes)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// The other half of "one gate over everything": a fact that matches better
|
||||||
|
// than the best note now answers, instead of losing to a note that only had
|
||||||
|
// to beat other notes.
|
||||||
|
t.Run("the better-matching fact answers", func(t *testing.T) {
|
||||||
|
h, _ := buildRecallHandler(t, q, []recallCase{
|
||||||
|
{text: "молоко стоит в холодильнике", score: 0.80, kind: "note"},
|
||||||
|
{text: "купил молоко в среду", score: 0.95, kind: "fact"},
|
||||||
|
})
|
||||||
|
if reply := askQuery(t, h, q); reply != "купил молоко в среду" {
|
||||||
|
t.Errorf("reply %q, want the fact read back", reply)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// The gate is untouched: two memories this close mean the embedder cannot
|
||||||
|
// tell them apart, and silence still beats a coin flip.
|
||||||
|
t.Run("no clear best stays silent", func(t *testing.T) {
|
||||||
|
h, _ := buildRecallHandler(t, q, []recallCase{
|
||||||
|
{text: "молоко стоит в холодильнике", score: 0.860, kind: "note"},
|
||||||
|
{text: "молоко закончилось", score: 0.858, kind: "note"},
|
||||||
|
})
|
||||||
|
if reply := askQuery(t, h, q); reply != "не знаю." {
|
||||||
|
t.Errorf("reply %q, want silence", reply)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
@@ -0,0 +1,144 @@
|
|||||||
|
// Quiet-mode toggle recognition — the pre-route keyword check that lets
|
||||||
|
// "тихий режим" flip the daemon-wide quiet_hours config without going through
|
||||||
|
// the router. Moved out of voice.go unchanged (Vikunja #321); the tests live in
|
||||||
|
// quiet_toggle_test.go.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"strings"
|
||||||
|
"unicode"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
)
|
||||||
|
|
||||||
|
// resolveQuietToggle — pre-route keyword check. Returns (reply, true) when
|
||||||
|
// the utterance is a quiet-on/off command; ("", false) otherwise. Called from
|
||||||
|
// runTurn BEFORE the router so a classifier miscue can't drop it — which means
|
||||||
|
// both the voice path and the text path (mavweb /api/chat, telegram) reach it,
|
||||||
|
// so a false positive here is a network-reachable way to flip a daemon-wide
|
||||||
|
// setting. See classifyQuietToggle for the matching rule.
|
||||||
|
func (h *reactiveHandler) resolveQuietToggle(ctx context.Context, text string) (string, bool) {
|
||||||
|
on, off := classifyQuietToggle(text)
|
||||||
|
if !on && !off {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
val := "false"
|
||||||
|
reply := "тихий режим выключен."
|
||||||
|
if on {
|
||||||
|
val = "true"
|
||||||
|
reply = "тихий режим включён. буду реже напоминать."
|
||||||
|
}
|
||||||
|
if _, err := h.api.WriteFact(ctx, ipc.WriteFactReq{
|
||||||
|
Ts: h.now(),
|
||||||
|
Kind: "config",
|
||||||
|
Key: "quiet_hours",
|
||||||
|
Value: val,
|
||||||
|
Source: "tap:voice",
|
||||||
|
Confidence: 1.0,
|
||||||
|
}); err != nil {
|
||||||
|
log.Printf("voice: write quiet_hours: %v", err)
|
||||||
|
return "не получилось переключить тихий режим.", true
|
||||||
|
}
|
||||||
|
return reply, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietInflections — the inflectional endings a stem may carry and still be
|
||||||
|
// the same word. Adjective/adverb/noun/verb endings, all ≤3 letters. This is
|
||||||
|
// what separates "тихий"/"тихом"/"тихо" (stem "тих" + a real ending) from
|
||||||
|
// "тихонько"/"потихоньку", which are different words: "онько" is not an
|
||||||
|
// ending, and "потихоньку" doesn't start with the stem at all.
|
||||||
|
var quietInflections = []string{
|
||||||
|
"", "а", "е", "и", "й", "о", "у", "ы", "ю", "я",
|
||||||
|
"ая", "ее", "ей", "ем", "ие", "ий", "им", "их", "ия", "ию", "ое", "ой", "ом", "ую", "ые", "ый", "ым", "ых", "ья",
|
||||||
|
"ами", "ого", "ому", "ыми", "ать", "ить", "ять",
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietStem reports whether tok is the given stem carrying at most one
|
||||||
|
// inflectional ending. Word boundaries come from tokenisation (see
|
||||||
|
// quietTokens), not from a regexp — Go's \b is ASCII-oriented and treats every
|
||||||
|
// Cyrillic letter as a non-word character, so `\bтих\b` would happily match
|
||||||
|
// inside "тихонько". Comparing whole tokens sidesteps that entirely.
|
||||||
|
func quietStem(tok, stem string) bool {
|
||||||
|
if !strings.HasPrefix(tok, stem) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
suffix := tok[len(stem):]
|
||||||
|
for _, e := range quietInflections {
|
||||||
|
if suffix == e {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietTokens splits an utterance into lowercase word tokens, dropping
|
||||||
|
// punctuation and spacing. Unicode-aware, so Cyrillic words tokenise the same
|
||||||
|
// way ASCII ones do.
|
||||||
|
func quietTokens(text string) []string {
|
||||||
|
return strings.FieldsFunc(strings.ToLower(strings.TrimSpace(text)), func(r rune) bool {
|
||||||
|
return !unicode.IsLetter(r) && !unicode.IsDigit(r)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietPhrase matches a pattern (a sequence of stems) against the token list.
|
||||||
|
// Multi-word patterns match any contiguous run of tokens — "включи тихий
|
||||||
|
// режим" carries "тихий режим". Single-word patterns match ONLY when they are
|
||||||
|
// the whole utterance: bare "тихо" is a command, but "в комнате тихо" is a
|
||||||
|
// remark about the room and must not flip a daemon-wide setting.
|
||||||
|
func quietPhrase(tokens, pattern []string) bool {
|
||||||
|
if len(pattern) == 0 || len(tokens) < len(pattern) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if len(pattern) == 1 {
|
||||||
|
return len(tokens) == 1 && quietStem(tokens[0], pattern[0])
|
||||||
|
}
|
||||||
|
for i := 0; i+len(pattern) <= len(tokens); i++ {
|
||||||
|
hit := true
|
||||||
|
for j, stem := range pattern {
|
||||||
|
if !quietStem(tokens[i+j], stem) {
|
||||||
|
hit = false
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if hit {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietOffPhrases / quietOnPhrases — the toggle vocabulary, as stem sequences.
|
||||||
|
var (
|
||||||
|
quietOffPhrases = [][]string{
|
||||||
|
{"quiet", "off"}, {"quiet", "end"},
|
||||||
|
{"громк", "режим"}, {"шумн", "режим"},
|
||||||
|
{"отмен", "тих"}, {"выключ", "тих"}, {"не", "тих"},
|
||||||
|
}
|
||||||
|
quietOnPhrases = [][]string{
|
||||||
|
{"quiet", "on"}, {"quiet", "mode"},
|
||||||
|
{"тих", "режим"}, {"не", "шум"}, {"не", "беспоко"},
|
||||||
|
{"тих"},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
// classifyQuietToggle reads an utterance as a quiet-mode command. OFF is
|
||||||
|
// resolved before ON for the same reason classifyConfirm checks negatives
|
||||||
|
// first: the OFF phrases are built out of the ON words ("выключи тихий"
|
||||||
|
// contains "тихий"), so scanning ON first would shadow them and "выключи
|
||||||
|
// тихий режим" would turn quiet mode on. Negation wins.
|
||||||
|
func classifyQuietToggle(text string) (on, off bool) {
|
||||||
|
tokens := quietTokens(text)
|
||||||
|
for _, p := range quietOffPhrases {
|
||||||
|
if quietPhrase(tokens, p) {
|
||||||
|
return false, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, p := range quietOnPhrases {
|
||||||
|
if quietPhrase(tokens, p) {
|
||||||
|
return true, false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false, false
|
||||||
|
}
|
||||||
@@ -0,0 +1,114 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
)
|
||||||
|
|
||||||
|
// quietFakeAPI records the WriteFact the toggle performs.
|
||||||
|
type quietFakeAPI struct {
|
||||||
|
ipc.UnimplementedCoreAPI
|
||||||
|
got ipc.WriteFactReq
|
||||||
|
call int
|
||||||
|
}
|
||||||
|
|
||||||
|
func (a *quietFakeAPI) WriteFact(_ context.Context, req ipc.WriteFactReq) (int64, error) {
|
||||||
|
a.got, a.call = req, a.call+1
|
||||||
|
return 1, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// quietVerdict — what a phrase should do to the setting.
|
||||||
|
type quietVerdict int
|
||||||
|
|
||||||
|
const (
|
||||||
|
quietNone quietVerdict = iota
|
||||||
|
quietOn
|
||||||
|
quietOff
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestResolveQuietToggle(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
text string
|
||||||
|
want quietVerdict
|
||||||
|
}{
|
||||||
|
// ON vocabulary.
|
||||||
|
{"quiet on", quietOn},
|
||||||
|
{"quiet mode", quietOn},
|
||||||
|
{"тихий режим", quietOn},
|
||||||
|
{"тихий", quietOn},
|
||||||
|
{"не шуми", quietOn},
|
||||||
|
{"не беспокоить", quietOn},
|
||||||
|
{"тихо", quietOn},
|
||||||
|
// ON, inflected / embedded in a sentence.
|
||||||
|
{"включи тихий режим", quietOn},
|
||||||
|
{"побудь в тихом режиме", quietOn},
|
||||||
|
{"Тихий Режим!", quietOn},
|
||||||
|
{"тихая", quietOn},
|
||||||
|
|
||||||
|
// OFF vocabulary — all seven, incl. the three that used to say ON.
|
||||||
|
{"quiet off", quietOff},
|
||||||
|
{"quiet end", quietOff},
|
||||||
|
{"громкий режим", quietOff},
|
||||||
|
{"шумный режим", quietOff},
|
||||||
|
{"отмени тихий", quietOff},
|
||||||
|
{"выключи тихий", quietOff},
|
||||||
|
{"не тихо", quietOff},
|
||||||
|
// OFF wins over the ON words it contains.
|
||||||
|
{"выключи тихий режим", quietOff},
|
||||||
|
{"отмени тихий режим пожалуйста", quietOff},
|
||||||
|
{"верни громкий режим", quietOff},
|
||||||
|
|
||||||
|
// False positives: "тихо"/"тихий" as ordinary Russian.
|
||||||
|
{"очень тихий сегодня день", quietNone},
|
||||||
|
{"в комнате тихо", quietNone},
|
||||||
|
{"тихонько напомни", quietNone},
|
||||||
|
{"потихоньку", quietNone},
|
||||||
|
{"тихонько", quietNone},
|
||||||
|
{"он говорил тихим голосом весь вечер", quietNone},
|
||||||
|
|
||||||
|
// Unrelated.
|
||||||
|
{"напомни завтра позвонить маме", quietNone},
|
||||||
|
{"какая погода", quietNone},
|
||||||
|
{"", quietNone},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.text, func(t *testing.T) {
|
||||||
|
api := &quietFakeAPI{}
|
||||||
|
h := &reactiveHandler{api: api, now: func() time.Time { return time.Unix(0, 0).UTC() }}
|
||||||
|
reply, handled := h.resolveQuietToggle(context.Background(), tc.text)
|
||||||
|
|
||||||
|
if tc.want == quietNone {
|
||||||
|
if handled || reply != "" {
|
||||||
|
t.Fatalf("%q: got (%q, %v), want no match", tc.text, reply, handled)
|
||||||
|
}
|
||||||
|
if api.call != 0 {
|
||||||
|
t.Fatalf("%q: wrote a fact on a non-match", tc.text)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if !handled {
|
||||||
|
t.Fatalf("%q: not handled, want %v", tc.text, tc.want)
|
||||||
|
}
|
||||||
|
wantReply, wantVal := "тихий режим выключен.", "false"
|
||||||
|
if tc.want == quietOn {
|
||||||
|
wantReply, wantVal = "тихий режим включён. буду реже напоминать.", "true"
|
||||||
|
}
|
||||||
|
if reply != wantReply {
|
||||||
|
t.Errorf("%q: reply = %q, want %q", tc.text, reply, wantReply)
|
||||||
|
}
|
||||||
|
if api.call != 1 {
|
||||||
|
t.Fatalf("%q: WriteFact called %d times, want 1", tc.text, api.call)
|
||||||
|
}
|
||||||
|
if api.got.Kind != "config" || api.got.Key != "quiet_hours" || api.got.Source != "tap:voice" || api.got.Confidence != 1.0 {
|
||||||
|
t.Errorf("%q: request shape = %+v", tc.text, api.got)
|
||||||
|
}
|
||||||
|
if api.got.Value != wantVal {
|
||||||
|
t.Errorf("%q: value = %q, want %q", tc.text, api.got.Value, wantVal)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
+18
-18
@@ -2,24 +2,24 @@ package main
|
|||||||
|
|
||||||
import "github.com/kami/maven/internal/memory"
|
import "github.com/kami/maven/internal/memory"
|
||||||
|
|
||||||
// bestRecall is the read side of the long-term memory store: the top hit's
|
// bestRecall is the read side of the long-term memory store: the top hit when
|
||||||
// stored text when it clears the confidence gate. This recalls across BOTH
|
// it clears the confidence gate. The index holds BOTH notes and facts, and
|
||||||
// notes and facts (facts aren't in the notes table, so this is the only path
|
// either can win — the caller looks at the returned hit's meta["type"] to see
|
||||||
// that can answer "when did I last …?" from a captured fact). A note hit here
|
// which. Facts aren't in the notes table, so this is the only path that can
|
||||||
// is redundant with the notes-RAG path — by design; the two indexes can diverge
|
// answer "when did I last …?" from a captured fact.
|
||||||
// once the backend is swapped for a persistent/external store. ok=false when
|
//
|
||||||
// there's no hit above the threshold or the hit carries no text.
|
// The whole hit is returned, not just its text, because "which memory answered"
|
||||||
func bestRecall(results []memory.Result, min float64) (string, bool) {
|
// decides how the answer is said: a note gets phrased in Maven's voice, a fact
|
||||||
if len(results) == 0 {
|
// is read back as stored.
|
||||||
return "", false
|
//
|
||||||
|
// ok=false when the hit fails the confidence gate (see memory.Confident: an
|
||||||
|
// absolute floor plus a margin over the runner-up) or carries no text.
|
||||||
|
func bestRecall(results []memory.Result, minScore, minMargin float64) (memory.Result, bool) {
|
||||||
|
if !memory.Confident(results, minScore, minMargin) {
|
||||||
|
return memory.Result{}, false
|
||||||
}
|
}
|
||||||
top := results[0]
|
if results[0].Meta["text"] == "" {
|
||||||
if top.Score < min {
|
return memory.Result{}, false
|
||||||
return "", false
|
|
||||||
}
|
}
|
||||||
text := top.Meta["text"]
|
return results[0], true
|
||||||
if text == "" {
|
|
||||||
return "", false
|
|
||||||
}
|
|
||||||
return text, true
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -8,23 +8,24 @@ import (
|
|||||||
|
|
||||||
func TestBestRecall(t *testing.T) {
|
func TestBestRecall(t *testing.T) {
|
||||||
const min = 0.55
|
const min = 0.55
|
||||||
|
const margin = 0.008
|
||||||
|
|
||||||
t.Run("empty results", func(t *testing.T) {
|
t.Run("empty results", func(t *testing.T) {
|
||||||
if _, ok := bestRecall(nil, min); ok {
|
if _, ok := bestRecall(nil, min, margin); ok {
|
||||||
t.Error("empty results returned ok")
|
t.Error("empty results returned ok")
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("top below threshold", func(t *testing.T) {
|
t.Run("top below threshold", func(t *testing.T) {
|
||||||
res := []memory.Result{{Score: 0.4, Meta: map[string]string{"text": "выпил воды"}}}
|
res := []memory.Result{{Score: 0.4, Meta: map[string]string{"text": "выпил воды"}}}
|
||||||
if _, ok := bestRecall(res, min); ok {
|
if _, ok := bestRecall(res, min, margin); ok {
|
||||||
t.Error("below-threshold hit returned ok")
|
t.Error("below-threshold hit returned ok")
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
|
|
||||||
t.Run("hit without text meta", func(t *testing.T) {
|
t.Run("hit without text meta", func(t *testing.T) {
|
||||||
res := []memory.Result{{Score: 0.9, Meta: map[string]string{"type": "fact"}}}
|
res := []memory.Result{{Score: 0.9, Meta: map[string]string{"type": "fact"}}}
|
||||||
if _, ok := bestRecall(res, min); ok {
|
if _, ok := bestRecall(res, min, margin); ok {
|
||||||
t.Error("textless hit returned ok")
|
t.Error("textless hit returned ok")
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
@@ -34,12 +35,43 @@ func TestBestRecall(t *testing.T) {
|
|||||||
{Score: 0.82, Meta: map[string]string{"text": "выпил воды в три часа", "type": "fact"}},
|
{Score: 0.82, Meta: map[string]string{"text": "выпил воды в три часа", "type": "fact"}},
|
||||||
{Score: 0.60, Meta: map[string]string{"text": "другое"}},
|
{Score: 0.60, Meta: map[string]string{"text": "другое"}},
|
||||||
}
|
}
|
||||||
got, ok := bestRecall(res, min)
|
got, ok := bestRecall(res, min, margin)
|
||||||
if !ok {
|
if !ok {
|
||||||
t.Fatal("clearing hit not returned")
|
t.Fatal("clearing hit not returned")
|
||||||
}
|
}
|
||||||
if got != "выпил воды в три часа" {
|
if got.Meta["text"] != "выпил воды в три часа" {
|
||||||
t.Errorf("wrong text: %q", got)
|
t.Errorf("wrong text: %q", got.Meta["text"])
|
||||||
|
}
|
||||||
|
if got.Meta["type"] != "fact" {
|
||||||
|
t.Errorf("kind lost: %q", got.Meta["type"])
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// The index holds notes and facts together, so a note has to be able to win
|
||||||
|
// it — for a long time it could not (Vikunja #373).
|
||||||
|
t.Run("a note can win", func(t *testing.T) {
|
||||||
|
res := []memory.Result{
|
||||||
|
{Score: 0.86, Meta: map[string]string{"text": "молоко в холодильнике", "type": "note"}},
|
||||||
|
{Score: 0.61, Meta: map[string]string{"text": "выпил воды", "type": "fact"}},
|
||||||
|
}
|
||||||
|
got, ok := bestRecall(res, min, margin)
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("clearly-best note not returned")
|
||||||
|
}
|
||||||
|
if got.Meta["type"] != "note" || got.Meta["text"] != "молоко в холодильнике" {
|
||||||
|
t.Errorf("got %v, want the note", got.Meta)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// The runner-up is almost as close, so the embedder cannot tell the two
|
||||||
|
// notes apart. Silence beats reading back a coin flip.
|
||||||
|
t.Run("runner-up too close", func(t *testing.T) {
|
||||||
|
res := []memory.Result{
|
||||||
|
{Score: 0.860, Meta: map[string]string{"text": "выпил воды в три часа"}},
|
||||||
|
{Score: 0.858, Meta: map[string]string{"text": "другое"}},
|
||||||
|
}
|
||||||
|
if _, ok := bestRecall(res, min, margin); ok {
|
||||||
|
t.Error("thin-margin hit returned ok")
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -7,6 +7,7 @@ import (
|
|||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/kami/maven/internal/llm"
|
"github.com/kami/maven/internal/llm"
|
||||||
|
"github.com/kami/maven/internal/persona"
|
||||||
"github.com/kami/maven/internal/router"
|
"github.com/kami/maven/internal/router"
|
||||||
"github.com/kami/maven/internal/voice"
|
"github.com/kami/maven/internal/voice"
|
||||||
)
|
)
|
||||||
@@ -23,13 +24,19 @@ type completer interface {
|
|||||||
type llmReplier struct {
|
type llmReplier struct {
|
||||||
c completer
|
c completer
|
||||||
stub *voice.StubReplier
|
stub *voice.StubReplier
|
||||||
|
|
||||||
|
// block renders the shared context block per turn (who he is, the time).
|
||||||
|
// nil ⇒ the prompt stands alone.
|
||||||
|
block func() string
|
||||||
}
|
}
|
||||||
|
|
||||||
func newLLMReplier(c completer) *llmReplier {
|
func newLLMReplier(c completer, block func() string) *llmReplier {
|
||||||
return &llmReplier{c: c, stub: voice.NewStubReplier()}
|
return &llmReplier{c: c, stub: voice.NewStubReplier(), block: block}
|
||||||
}
|
}
|
||||||
|
|
||||||
const replySystem = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), тепло и по-русски. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Respond ONLY with valid JSON: {"response": "...", "mood": "neutral"}.`
|
const replySystem = `Ты — Maven, домашняя ассистентка (о себе — в женском роде). Владелец — мужчина, говоришь с ним на "ты", в единственном числе; никогда не "вы"/"ваш" и не "он"/"его". Подтверди действие РОВНО ОДНИМ коротким предложением (≤120 символов), по-русски, спокойно и без официальных формулировок. Не задавай вопросов, не повторяй слова, не добавляй ничего после точки. Отвечай ТОЛЬКО одним объектом JSON с полями "response" (текст) и "mood" (ровно одно из: neutral, happy, thinking, tired, confused).
|
||||||
|
Пример: {"response": "Записала, что ты выпил стакан воды.", "mood": "neutral"}
|
||||||
|
Никогда не пиши "..." в поле response.`
|
||||||
|
|
||||||
func (r *llmReplier) Reply(d router.Decision) string {
|
func (r *llmReplier) Reply(d router.Decision) string {
|
||||||
if d.Clarify {
|
if d.Clarify {
|
||||||
@@ -38,7 +45,7 @@ func (r *llmReplier) Reply(d router.Decision) string {
|
|||||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||||
defer cancel()
|
defer cancel()
|
||||||
out, err := r.c.Complete(ctx, llm.Req{
|
out, err := r.c.Complete(ctx, llm.Req{
|
||||||
System: replySystem,
|
System: persona.Prepend(r.block, replySystem),
|
||||||
User: replyContext(d),
|
User: replyContext(d),
|
||||||
MaxTokens: 512,
|
MaxTokens: 512,
|
||||||
RepeatPenalty: 1.3,
|
RepeatPenalty: 1.3,
|
||||||
|
|||||||
@@ -9,12 +9,15 @@ import (
|
|||||||
"github.com/kami/maven/internal/voice"
|
"github.com/kami/maven/internal/voice"
|
||||||
)
|
)
|
||||||
|
|
||||||
type mockCompleter struct{ out string; err error }
|
type mockCompleter struct {
|
||||||
|
out string
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
|
||||||
func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
|
func (m mockCompleter) Complete(_ context.Context, _ llm.Req) (string, error) { return m.out, m.err }
|
||||||
|
|
||||||
func TestLLMReplierReturnsLLMReply(t *testing.T) {
|
func TestLLMReplierReturnsLLMReply(t *testing.T) {
|
||||||
r := newLLMReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`})
|
r := newLLMReplier(mockCompleter{out: `{"response":"записала, кофе закончился","mood":"neutral"}`}, nil)
|
||||||
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||||
if got != "записала, кофе закончился" {
|
if got != "записала, кофе закончился" {
|
||||||
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
||||||
@@ -22,7 +25,7 @@ func TestLLMReplierReturnsLLMReply(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestLLMReplierFallsBackToPlainText(t *testing.T) {
|
func TestLLMReplierFallsBackToPlainText(t *testing.T) {
|
||||||
r := newLLMReplier(mockCompleter{out: "записала, кофе закончился"})
|
r := newLLMReplier(mockCompleter{out: "записала, кофе закончился"}, nil)
|
||||||
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
got := r.Reply(router.Decision{Intent: router.IntentNote, Slots: router.Slots{Text: "кофе закончился"}})
|
||||||
if got != "записала, кофе закончился" {
|
if got != "записала, кофе закончился" {
|
||||||
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
t.Errorf("got %q, want %q", got, "записала, кофе закончился")
|
||||||
@@ -30,7 +33,7 @@ func TestLLMReplierFallsBackToPlainText(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestLLMReplierFallsBackToStubOnError(t *testing.T) {
|
func TestLLMReplierFallsBackToStubOnError(t *testing.T) {
|
||||||
r := newLLMReplier(mockCompleter{err: errTestLLMDown})
|
r := newLLMReplier(mockCompleter{err: errTestLLMDown}, nil)
|
||||||
noteDec := router.Decision{Intent: router.IntentNote}
|
noteDec := router.Decision{Intent: router.IntentNote}
|
||||||
got := r.Reply(noteDec)
|
got := r.Reply(noteDec)
|
||||||
want := voice.NewStubReplier().Reply(noteDec)
|
want := voice.NewStubReplier().Reply(noteDec)
|
||||||
@@ -40,7 +43,7 @@ func TestLLMReplierFallsBackToStubOnError(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestLLMReplierFallsBackToStubOnEmpty(t *testing.T) {
|
func TestLLMReplierFallsBackToStubOnEmpty(t *testing.T) {
|
||||||
r := newLLMReplier(mockCompleter{out: ""})
|
r := newLLMReplier(mockCompleter{out: ""}, nil)
|
||||||
noteDec := router.Decision{Intent: router.IntentNote}
|
noteDec := router.Decision{Intent: router.IntentNote}
|
||||||
got := r.Reply(noteDec)
|
got := r.Reply(noteDec)
|
||||||
want := voice.NewStubReplier().Reply(noteDec)
|
want := voice.NewStubReplier().Reply(noteDec)
|
||||||
@@ -50,7 +53,7 @@ func TestLLMReplierFallsBackToStubOnEmpty(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestLLMReplierClarifyUsesStub(t *testing.T) {
|
func TestLLMReplierClarifyUsesStub(t *testing.T) {
|
||||||
r := newLLMReplier(mockCompleter{out: "я всё поняла"})
|
r := newLLMReplier(mockCompleter{out: "я всё поняла"}, nil)
|
||||||
clarifyDec := router.Decision{Clarify: true}
|
clarifyDec := router.Decision{Clarify: true}
|
||||||
got := r.Reply(clarifyDec)
|
got := r.Reply(clarifyDec)
|
||||||
want := voice.NewStubReplier().Reply(clarifyDec)
|
want := voice.NewStubReplier().Reply(clarifyDec)
|
||||||
|
|||||||
@@ -0,0 +1,180 @@
|
|||||||
|
// Package main — ruwords.go holds Russian language + calendar/time formatting
|
||||||
|
// helpers used by the voice reply paths (replySystem, the reminder/routine
|
||||||
|
// phrasing, etc). Pure functions, no receivers: weekday/month name tables,
|
||||||
|
// plural agreement, clock/date rendering, and the "do I actually know this
|
||||||
|
// place/day" guards that pick an honest reply over a confidently wrong one.
|
||||||
|
// Extend this file rather than voice.go for anything in that shape.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
var ruWeekdays = []string{
|
||||||
|
"воскресенье", "понедельник", "вторник", "среда",
|
||||||
|
"четверг", "пятница", "суббота",
|
||||||
|
}
|
||||||
|
|
||||||
|
var ruMonths = []string{
|
||||||
|
"января", "февраля", "марта", "апреля", "мая", "июня",
|
||||||
|
"июля", "августа", "сентября", "октября", "ноября", "декабря",
|
||||||
|
}
|
||||||
|
|
||||||
|
// onlyLocalTimeReply — the honest answer when the user asks the time somewhere
|
||||||
|
// other than here. She only keeps one clock, and saying so is better than
|
||||||
|
// naming the wrong city's time.
|
||||||
|
//
|
||||||
|
// There used to be a city→time-zone table here. It was removed on purpose: the
|
||||||
|
// user only ever asks for local time, so the table was a second list of cities
|
||||||
|
// to keep in step with the weather one for no gain.
|
||||||
|
const onlyLocalTimeReply = "я знаю только местное время, про другие города пока не скажу."
|
||||||
|
|
||||||
|
// notPlaceAfterV — words that follow "в" without naming a place, so
|
||||||
|
// mentionsUnknownPlace does not mistake them for a city.
|
||||||
|
var notPlaceAfterV = map[string]bool{
|
||||||
|
"данный": true, "данную": true, "этот": true, "эту": true,
|
||||||
|
"котором": true, "какое": true, "какой": true, "который": true,
|
||||||
|
"общем": true, "точности": true, "курсе": true, "сутках": true,
|
||||||
|
"часах": true, "минутах": true, "секундах": true, "неделе": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// mentionsUnknownPlace reports whether the question has a "в <слово>" phrase
|
||||||
|
// that looks like a place we do not know ("который час в киеве"). Used only to
|
||||||
|
// pick the honest "local time only" reply instead of answering local time as
|
||||||
|
// if it were the city's.
|
||||||
|
func mentionsUnknownPlace(u string) bool {
|
||||||
|
toks := strings.Fields(u)
|
||||||
|
for i := 0; i+1 < len(toks); i++ {
|
||||||
|
if toks[i] != "в" && toks[i] != "во" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
next := strings.Trim(toks[i+1], ".,?!")
|
||||||
|
if next == "" || notPlaceAfterV[next] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// A number after "в" is a clock ("в 5 часов"), not a place.
|
||||||
|
if _, err := strconv.Atoi(strings.SplitN(next, ":", 2)[0]); err == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// onlyNearDaysReply — she can work out today, tomorrow, the day after and
|
||||||
|
// yesterday, and nothing further. Said out loud instead of answering today's
|
||||||
|
// date for a day she did not understand.
|
||||||
|
const onlyNearDaysReply = "я считаю только сегодня, завтра, послезавтра и вчера — про другие дни пока не скажу."
|
||||||
|
|
||||||
|
// dayWords — day references the calendar parser cannot resolve. A weekday name
|
||||||
|
// or a "через …" phrase means he asked about a specific other day.
|
||||||
|
var dayWords = []string{
|
||||||
|
"понедельник", "вторник", "сред", "четверг", "пятниц", "суббот", "воскресен",
|
||||||
|
"через", "monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday",
|
||||||
|
}
|
||||||
|
|
||||||
|
// mentionsUnknownDay reports whether the question names a day the calendar
|
||||||
|
// parser could not resolve. Mirror of mentionsUnknownPlace: it exists only to
|
||||||
|
// pick an honest reply over a confidently wrong one.
|
||||||
|
//
|
||||||
|
// Only called after ParseCalendarDate has already failed, so "завтра" and the
|
||||||
|
// other words it does know never reach here.
|
||||||
|
func mentionsUnknownDay(u string) bool {
|
||||||
|
for _, w := range dayWords {
|
||||||
|
if strings.Contains(u, w) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// ruClock renders the clock part of the time reply: "15 часов 4 минуты".
|
||||||
|
func ruClock(t time.Time) string {
|
||||||
|
h, m := t.Hour(), t.Minute()
|
||||||
|
hourWord := ruPlural(h, "час", "часа", "часов")
|
||||||
|
if m == 0 {
|
||||||
|
return fmt.Sprintf("%d %s ровно", h, hourWord)
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("%d %s %d %s", h, hourWord, m, ruPlural(m, "минута", "минуты", "минут"))
|
||||||
|
}
|
||||||
|
|
||||||
|
// dayPrefix names the day relative to now ("завтра", "вчера", …) so the date
|
||||||
|
// reply opens the way a person would say it.
|
||||||
|
func dayPrefix(now, day time.Time) string {
|
||||||
|
base := time.Date(now.Year(), now.Month(), now.Day(), 0, 0, 0, 0, now.Location())
|
||||||
|
switch int(day.Sub(base).Hours() / 24) {
|
||||||
|
case -1:
|
||||||
|
return "вчера"
|
||||||
|
case 0:
|
||||||
|
return "сегодня"
|
||||||
|
case 1:
|
||||||
|
return "завтра"
|
||||||
|
case 2:
|
||||||
|
return "послезавтра"
|
||||||
|
}
|
||||||
|
return "это"
|
||||||
|
}
|
||||||
|
|
||||||
|
func ruPlural(n int, one, two, many string) string {
|
||||||
|
n = n % 100
|
||||||
|
if n > 10 && n < 20 {
|
||||||
|
return many
|
||||||
|
}
|
||||||
|
n = n % 10
|
||||||
|
switch n {
|
||||||
|
case 1:
|
||||||
|
return one
|
||||||
|
case 2, 3, 4:
|
||||||
|
return two
|
||||||
|
default:
|
||||||
|
return many
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// hasDurationWords checks whether u is asking about elapsed/remaining time
|
||||||
|
// rather than the current clock — guards replySystem from replying "сейчас
|
||||||
|
// X часов" to "сколько времени прошло". Mirrors the stage0.go build filter.
|
||||||
|
func hasDurationWords(u string) bool {
|
||||||
|
s := strings.ToLower(strings.TrimSpace(u))
|
||||||
|
// First-word duration markers (same keywords as timeQueryBuild in stage0).
|
||||||
|
first := strings.Fields(s)
|
||||||
|
if len(first) > 0 {
|
||||||
|
switch first[0] {
|
||||||
|
case "прошло", "осталось", "пройдет", "минуло", "проходит":
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Broader duration keywords appearing anywhere in the utterance.
|
||||||
|
if strings.Contains(s, "прошло") || strings.Contains(s, "осталось") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
if strings.Contains(s, " до ") {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// formatTime returns a human-readable Russian time string for a fact timestamp.
|
||||||
|
// Used by the query handler when answering "когда я это сделал?"-style questions.
|
||||||
|
func formatTime(t time.Time) string {
|
||||||
|
now := time.Now()
|
||||||
|
if t.After(now.Add(-2*time.Minute)) && t.Before(now.Add(2*time.Minute)) {
|
||||||
|
return "только что"
|
||||||
|
}
|
||||||
|
diff := now.Sub(t)
|
||||||
|
switch {
|
||||||
|
case diff < 10*time.Minute:
|
||||||
|
return "несколько минут назад"
|
||||||
|
case diff < 60*time.Minute:
|
||||||
|
return fmt.Sprintf("%d минут назад", int(diff.Minutes()))
|
||||||
|
case diff < 2*time.Hour:
|
||||||
|
return "час назад"
|
||||||
|
case diff < 24*time.Hour:
|
||||||
|
return fmt.Sprintf("%d часа назад", int(diff.Hours()))
|
||||||
|
default:
|
||||||
|
return t.Format("2 января 15:04")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,85 @@
|
|||||||
|
// Package main — strutil.go holds small, receiver-free string utilities used
|
||||||
|
// across the voice reply paths: trimming a wake token, pulling out the first
|
||||||
|
// word or first line, and a minimal JSON string encoder for the one payload
|
||||||
|
// shape that needs it. Extend this file rather than voice.go for anything in
|
||||||
|
// that shape.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// stripWake removes a leading wake token (any script the STT phonetically
|
||||||
|
// transcribes "Maven" as) so the verb is the first word.
|
||||||
|
func stripWake(u string) string {
|
||||||
|
stripped, had := router.StripWakeToken(u)
|
||||||
|
if !had {
|
||||||
|
return strings.TrimSpace(u)
|
||||||
|
}
|
||||||
|
return stripped
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstWord returns the first whitespace-delimited token (lowercased) — the
|
||||||
|
// proposed tool's name.
|
||||||
|
func firstWord(s string) string {
|
||||||
|
f := strings.Fields(s)
|
||||||
|
if len(f) == 0 {
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
return strings.ToLower(f[0])
|
||||||
|
}
|
||||||
|
|
||||||
|
// firstLine — the first non-empty line of a tool's output, for a short spoken
|
||||||
|
// reply (the full output goes to the log, not the TTS). Trimmed to keep the
|
||||||
|
// utterance sane if a command dumps a wall of text.
|
||||||
|
func firstLine(s string) string {
|
||||||
|
for _, line := range strings.Split(s, "\n") {
|
||||||
|
line = strings.TrimSpace(line)
|
||||||
|
if line != "" {
|
||||||
|
if len(line) > 200 {
|
||||||
|
line = line[:200]
|
||||||
|
}
|
||||||
|
return line
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// jsonString — a one-line JSON string encoder without dragging encoding/json
|
||||||
|
// into the top of this file. Used to wrap a reminder payload's text field;
|
||||||
|
// the router's reminder Slots are already absolute (DateTimeParser resolved
|
||||||
|
// relative→absolute), the payload shape is conventional {"text":...}.
|
||||||
|
func jsonString(s string) string {
|
||||||
|
// minimal JSON string escape — quotes + backslash + control chars.
|
||||||
|
// adequate for the reminder payload's text field; not a general JSON
|
||||||
|
// encoder. The chroma / RAG modules (when they land) use a real json
|
||||||
|
// encoder for richer payloads. Keep it inline here so the import
|
||||||
|
// direction stays narrow.
|
||||||
|
var b []byte
|
||||||
|
b = append(b, '"')
|
||||||
|
for _, r := range s {
|
||||||
|
switch r {
|
||||||
|
case '"':
|
||||||
|
b = append(b, '\\', '"')
|
||||||
|
case '\\':
|
||||||
|
b = append(b, '\\', '\\')
|
||||||
|
case '\n':
|
||||||
|
b = append(b, '\\', 'n')
|
||||||
|
case '\r':
|
||||||
|
b = append(b, '\\', 'r')
|
||||||
|
case '\t':
|
||||||
|
b = append(b, '\\', 't')
|
||||||
|
default:
|
||||||
|
if r < 0x20 {
|
||||||
|
b = append(b, []byte(fmt.Sprintf("\\u%04x", r))...)
|
||||||
|
} else {
|
||||||
|
b = append(b, []byte(string(r))...)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b = append(b, '"')
|
||||||
|
return string(b)
|
||||||
|
}
|
||||||
@@ -0,0 +1,78 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
)
|
||||||
|
|
||||||
|
// systemHandler — a handler with nothing but a fixed clock, which is all
|
||||||
|
// replySystem needs.
|
||||||
|
func systemHandler(now time.Time) *reactiveHandler {
|
||||||
|
return &reactiveHandler{now: func() time.Time { return now }}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestReplySystemDateOffset — "какое число завтра" must answer tomorrow's
|
||||||
|
// date, not today's (Vikunja #388).
|
||||||
|
func TestReplySystemDateOffset(t *testing.T) {
|
||||||
|
// Thursday, 30 July 2026.
|
||||||
|
now := time.Date(2026, 7, 30, 14, 5, 0, 0, time.UTC)
|
||||||
|
h := systemHandler(now)
|
||||||
|
cases := []struct{ utterance, want string }{
|
||||||
|
{"какое сегодня число", "сегодня четверг, 30 июля 2026 года"},
|
||||||
|
{"какое число", "сегодня четверг, 30 июля 2026 года"},
|
||||||
|
{"какое число завтра", "завтра пятница, 31 июля 2026 года"},
|
||||||
|
{"какое число послезавтра", "послезавтра суббота, 1 августа 2026 года"},
|
||||||
|
{"какое было число вчера", "вчера среда, 29 июля 2026 года"},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
got := h.replySystem(context.Background(), router.Decision{Utterance: c.utterance})
|
||||||
|
if got != c.want {
|
||||||
|
t.Errorf("replySystem(%q) = %q, want %q", c.utterance, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A day she cannot work out must not come back as today's date — that is the
|
||||||
|
// same silent wrong answer #388 was about, one step further out.
|
||||||
|
func TestReplySystemUnknownDayIsHonest(t *testing.T) {
|
||||||
|
now := time.Date(2026, 7, 30, 14, 5, 0, 0, time.UTC)
|
||||||
|
h := systemHandler(now)
|
||||||
|
for _, u := range []string{
|
||||||
|
"какое число в пятницу",
|
||||||
|
"какое число через неделю",
|
||||||
|
"какое число в понедельник",
|
||||||
|
} {
|
||||||
|
got := h.replySystem(context.Background(), router.Decision{Utterance: u})
|
||||||
|
if got != onlyNearDaysReply {
|
||||||
|
t.Errorf("replySystem(%q) = %q, want the honest reply", u, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The days she does know must not be caught by the same guard.
|
||||||
|
if got := h.replySystem(context.Background(), router.Decision{Utterance: "какое число завтра"}); got == onlyNearDaysReply {
|
||||||
|
t.Error("завтра was treated as an unknown day")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestReplySystemClockCity — the clock arm must not answer local time for a
|
||||||
|
// question about another city (Vikunja #388). She keeps one clock, so every
|
||||||
|
// named place gets the honest "local time only" answer.
|
||||||
|
func TestReplySystemClockCity(t *testing.T) {
|
||||||
|
now := time.Date(2026, 7, 30, 12, 0, 0, 0, time.UTC)
|
||||||
|
h := systemHandler(now)
|
||||||
|
cases := []struct{ utterance, want string }{
|
||||||
|
{"который час", "сейчас 12 часов ровно"},
|
||||||
|
{"который час в киеве", onlyLocalTimeReply},
|
||||||
|
{"сколько времени в москве", onlyLocalTimeReply},
|
||||||
|
{"который час в лондоне", onlyLocalTimeReply},
|
||||||
|
{"который час в бишкеке", onlyLocalTimeReply},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
got := h.replySystem(context.Background(), router.Decision{Utterance: c.utterance})
|
||||||
|
if got != c.want {
|
||||||
|
t.Errorf("replySystem(%q) = %q, want %q", c.utterance, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -24,6 +24,7 @@ import (
|
|||||||
"github.com/kami/maven/internal/ipc"
|
"github.com/kami/maven/internal/ipc"
|
||||||
"github.com/kami/maven/internal/loop"
|
"github.com/kami/maven/internal/loop"
|
||||||
"github.com/kami/maven/internal/morning"
|
"github.com/kami/maven/internal/morning"
|
||||||
|
"github.com/kami/maven/internal/pattern"
|
||||||
"github.com/kami/maven/internal/phraser"
|
"github.com/kami/maven/internal/phraser"
|
||||||
"github.com/kami/maven/internal/routine"
|
"github.com/kami/maven/internal/routine"
|
||||||
"github.com/kami/maven/internal/store"
|
"github.com/kami/maven/internal/store"
|
||||||
@@ -67,6 +68,14 @@ type tickLoop struct {
|
|||||||
morningRoutines []morning.Routine
|
morningRoutines []morning.Routine
|
||||||
morningLast map[string]time.Time
|
morningLast map[string]time.Time
|
||||||
|
|
||||||
|
// proposalCfg — announcement policy for routines the tick inferred itself.
|
||||||
|
// nil ⇒ detect silently, never announce (the default). lastProposalAt is
|
||||||
|
// the cooldown clock, in-memory on purpose: a restart is allowed to permit
|
||||||
|
// one more announcement, and a restart-per-day loop is a bigger problem
|
||||||
|
// than a duplicate proposal notice.
|
||||||
|
proposalCfg *config.PatternProposalConfig
|
||||||
|
lastProposalAt time.Time
|
||||||
|
|
||||||
// digestQ — in-memory queue of eligible nudges waiting for batch flush.
|
// digestQ — in-memory queue of eligible nudges waiting for batch flush.
|
||||||
// populated when digestCfg != nil && digestCfg.Enabled.
|
// populated when digestCfg != nil && digestCfg.Enabled.
|
||||||
digestQ []QueuedNudge
|
digestQ []QueuedNudge
|
||||||
@@ -92,6 +101,7 @@ func newTickLoop(
|
|||||||
digestCfg *config.DigestConfig,
|
digestCfg *config.DigestConfig,
|
||||||
routines []routine.Routine,
|
routines []routine.Routine,
|
||||||
morningRoutines []morning.Routine,
|
morningRoutines []morning.Routine,
|
||||||
|
proposalCfg *config.PatternProposalConfig,
|
||||||
) *tickLoop {
|
) *tickLoop {
|
||||||
return &tickLoop{
|
return &tickLoop{
|
||||||
store: st,
|
store: st,
|
||||||
@@ -108,6 +118,7 @@ func newTickLoop(
|
|||||||
routineLast: make(map[string]time.Time),
|
routineLast: make(map[string]time.Time),
|
||||||
morningRoutines: morningRoutines,
|
morningRoutines: morningRoutines,
|
||||||
morningLast: make(map[string]time.Time),
|
morningLast: make(map[string]time.Time),
|
||||||
|
proposalCfg: proposalCfg,
|
||||||
lastPhrase: make(map[string]delivery.PhrasedNudge),
|
lastPhrase: make(map[string]delivery.PhrasedNudge),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -180,17 +191,40 @@ func (t *tickLoop) tick(ctx context.Context, now time.Time) {
|
|||||||
// (with dedup) avoids re-queueing the same rule after a flush.
|
// (with dedup) avoids re-queueing the same rule after a flush.
|
||||||
t.maybeFlush(ctx, now, state)
|
t.maybeFlush(ctx, now, state)
|
||||||
|
|
||||||
|
// gate-suppressed digest (Vikunja #281): rules the restraint gate held
|
||||||
|
// back this tick (quiet hours / away / calendar-busy), not because they
|
||||||
|
// weren't due, but because it wasn't the moment. Some of those are worth
|
||||||
|
// resurfacing later instead of just being lost — loop.DigestEligible
|
||||||
|
// draws that line. This is a SEPARATE mechanism from the in-memory
|
||||||
|
// digestQ above: that one batches candidates the gate already ALLOWED to
|
||||||
|
// fire; this one durably holds candidates the gate BLOCKED.
|
||||||
|
t.enqueueSuppressedDigest(ctx, trace, state, now)
|
||||||
|
t.expireStaleDigest(ctx, now)
|
||||||
|
t.maybeDrainDigest(ctx, state, now)
|
||||||
|
|
||||||
// routines: operator-declared scheduled behaviors. fire the ones whose cron
|
// routines: operator-declared scheduled behaviors. fire the ones whose cron
|
||||||
// crossed since last fire, delivered through the normal routing (voice when
|
// crossed since last fire, delivered through the normal routing (voice when
|
||||||
// present, away channels otherwise). bodies are literal operator text — not
|
// present, away channels otherwise). bodies are literal operator text — not
|
||||||
// LLM-phrased — so a routine can't hallucinate. severity comes from config.
|
// LLM-phrased — so a routine can't hallucinate. severity comes from config.
|
||||||
t.fireRoutines(ctx, now, state)
|
t.fireRoutines(ctx, now, state)
|
||||||
|
|
||||||
|
// accepted routines: patterns the user confirmed. read straight from the
|
||||||
|
// store each tick so the schedule survives a restart.
|
||||||
|
t.fireAcceptedRoutines(ctx, now, state)
|
||||||
|
|
||||||
// morning routines: daily checklists (medicine/water/pets/...), nagged at
|
// morning routines: daily checklists (medicine/water/pets/...), nagged at
|
||||||
// most once per day per routine, and only for items still unevidenced at
|
// most once per day per routine, and only for items still unevidenced at
|
||||||
// nudge time. See internal/morning for the "why not four timers" rationale.
|
// nudge time. See internal/morning for the "why not four timers" rationale.
|
||||||
t.fireMorningRoutines(ctx, now, state)
|
t.fireMorningRoutines(ctx, now, state)
|
||||||
|
|
||||||
|
// pattern detection: scan every action+object pair with recorded events
|
||||||
|
// and propose a routine for any stable one not already decided (Vikunja
|
||||||
|
// #43). This used to only run as a side effect of the voice fact-write
|
||||||
|
// path, so a pattern already sitting in history went unnoticed until he
|
||||||
|
// happened to mention it again by voice. See patterns.go and
|
||||||
|
// detectPatterns below for how idempotence and dismissal are respected.
|
||||||
|
t.detectPatterns(ctx, now, state)
|
||||||
|
|
||||||
// reminders: gate-bypassing class. fired once, marked after a successful
|
// reminders: gate-bypassing class. fired once, marked after a successful
|
||||||
// delivery. a failed send leaves the reminder pending — the next tick
|
// delivery. a failed send leaves the reminder pending — the next tick
|
||||||
// re-gathers and re-attempts.
|
// re-gathers and re-attempts.
|
||||||
@@ -338,6 +372,235 @@ func (t *tickLoop) flushDigest(ctx context.Context, now time.Time, state loop.St
|
|||||||
t.digestQ = nil
|
t.digestQ = nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// detectPatterns runs the pattern detector proactively over every
|
||||||
|
// action+object pair that has ever produced an event, independent of
|
||||||
|
// whichever fact write (or channel) last touched it (Vikunja #43). This is
|
||||||
|
// what makes pattern inference actually proactive: it fires on the daemon's
|
||||||
|
// own schedule reading accumulated history, not only as a side effect of a
|
||||||
|
// live voice turn.
|
||||||
|
//
|
||||||
|
// Idempotence and noise are handled by the store, not here — this function
|
||||||
|
// is safe to call every tick:
|
||||||
|
// - Same pattern, tick after tick: detectAndPropose's LookupProposedRoutine
|
||||||
|
// check plus proposed_routines' UNIQUE(action, object) constraint (with
|
||||||
|
// CreateProposedRoutine's ON CONFLICT DO NOTHING) mean a pair that
|
||||||
|
// already has a row — in ANY status — produces no second row and no log
|
||||||
|
// spam beyond the one line at genuine creation.
|
||||||
|
// - A DISMISSED proposal must never come back. DismissProposedRoutine flips
|
||||||
|
// status in place; the row is never deleted. So the same Lookup check
|
||||||
|
// that stops a duplicate "proposed" also stops a "dismissed" one from
|
||||||
|
// resurrecting — there is nothing tick-specific to get right here beyond
|
||||||
|
// calling the same shared path the voice route already used.
|
||||||
|
//
|
||||||
|
// By default this only creates a row for the /routines page to show: it does
|
||||||
|
// not notify, ring, or speak. Detection is not the same act as disturbing him
|
||||||
|
// about it, and Maven is "not a nag, not autonomous" (CLAUDE.md). Announcing
|
||||||
|
// is opt-in through the pattern_proposals config block — see announceProposal
|
||||||
|
// for the restraints that apply even then. A proposal only starts producing
|
||||||
|
// recurring nudges once he accepts it (fireAcceptedRoutines).
|
||||||
|
func (t *tickLoop) detectPatterns(ctx context.Context, now time.Time, state loop.State) {
|
||||||
|
pairs, err := t.store.DistinctEventPairs(ctx)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: distinct event pairs: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
announced := false
|
||||||
|
for _, p := range pairs {
|
||||||
|
r, _, err := detectAndPropose(ctx, t.store, p.Action, p.Object, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: detect pattern %s/%s: %v", p.Action, p.Object, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if r == nil {
|
||||||
|
continue // no stable pattern, or already proposed/accepted/dismissed
|
||||||
|
}
|
||||||
|
log.Printf("tick: proposed routine: %s/%s every %.1f days", r.Action, r.Object, r.IntervalDays)
|
||||||
|
// One announcement per tick at most, whatever the scan turned up. The
|
||||||
|
// rest are on /routines; they are not lost, they are just not shouted.
|
||||||
|
if announced {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
announced = t.announceProposal(ctx, r, now, state)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// announceProposal offers a freshly inferred routine through the ordinary
|
||||||
|
// care-delivery path, if announcing is switched on at all. Returns true when
|
||||||
|
// something was actually sent.
|
||||||
|
//
|
||||||
|
// Everything here is restraint. The feature is off unless configured; when on
|
||||||
|
// it is sev1 (the lowest severity, so quiet hours, away presence and snooze
|
||||||
|
// all suppress it via loop.Gate exactly like a care nudge); it is spaced by
|
||||||
|
// proposalCfg.Cooldown across every pair, not per pair; and a suppressed or
|
||||||
|
// dropped announcement is NOT retried — the cooldown clock advances only on a
|
||||||
|
// real send, but the proposal row already exists, so the next tick will not
|
||||||
|
// re-detect it and nothing queues up behind it. A missed announcement means
|
||||||
|
// he reads it on /routines instead, which is the whole point of the page.
|
||||||
|
//
|
||||||
|
// The body is the detector's own literal Russian phrasing (pattern.PhraseRoutine
|
||||||
|
// — "ты заправляешь поилку раз в 7 дней — напоминать?"), not LLM-generated, so
|
||||||
|
// an inferred routine cannot arrive worded as something Maven never observed.
|
||||||
|
func (t *tickLoop) announceProposal(ctx context.Context, r *pattern.ProposedRoutine, now time.Time, state loop.State) bool {
|
||||||
|
if !t.proposalCfg.AnnounceProposals() {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
cooldown := time.Duration(t.proposalCfg.Cooldown)
|
||||||
|
if cooldown <= 0 {
|
||||||
|
cooldown = config.DefaultProposalCooldown
|
||||||
|
}
|
||||||
|
if !t.lastProposalAt.IsZero() && now.Sub(t.lastProposalAt) < cooldown {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
rule := loop.Rule{Name: "proposal:" + r.Action + " " + r.Object, Severity: loop.Sev1}
|
||||||
|
if !loop.Gate(state, rule) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
body := pattern.PhraseRoutine(r)
|
||||||
|
pn := delivery.PhrasedNudge{
|
||||||
|
Candidate: loop.Candidate{Rule: rule, Severity: rule.Severity, State: state},
|
||||||
|
Body: body,
|
||||||
|
Summary: body,
|
||||||
|
}
|
||||||
|
sent, err := t.dispatcher.DispatchNudge(ctx, pn, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: announce proposal %s/%s: %v", r.Action, r.Object, err)
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if len(sent) == 0 {
|
||||||
|
return false // routing dropped it — /routines still has it.
|
||||||
|
}
|
||||||
|
t.lastProposalAt = now
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
// digestExpiry — how long a gate-suppressed care nudge stays worth
|
||||||
|
// resurfacing. 24h: these are daily-cadence rules (water/meal/break run on
|
||||||
|
// hour-scale cooldowns and re-derive from facts that reset every day), so a
|
||||||
|
// digest entry that outlives one full day is describing a day that's already
|
||||||
|
// over — "you skipped a break yesterday" said tomorrow evening is noise, not
|
||||||
|
// news. Bounding at one day also means a digest can never silently span a
|
||||||
|
// weekend of quiet hours into an unbounded backlog.
|
||||||
|
const digestExpiry = 24 * time.Hour
|
||||||
|
|
||||||
|
// maxDigestSpokenItems — the bundle read-out is capped so "batched, not
|
||||||
|
// dropped" cannot regress into "she dumps twelve things on me the moment I
|
||||||
|
// walk in" — a digest that nags in bulk is worse than the drops it replaced.
|
||||||
|
// Anything beyond the cap is still marked drained (it did get its moment;
|
||||||
|
// the cap limits WORDS, not whether it counted) and folded into a trailing
|
||||||
|
// count instead of being spoken in full.
|
||||||
|
const maxDigestSpokenItems = 3
|
||||||
|
|
||||||
|
// enqueueSuppressedDigest scans this tick's trace for care candidates the
|
||||||
|
// gate blocked for a genuine restraint reason and durably records the
|
||||||
|
// digest-eligible ones (loop.DigestEligible). Phrasing happens once, here,
|
||||||
|
// at enqueue time — not re-derived at drain time — the same way queueNudge
|
||||||
|
// phrases once and caches, so a rule suppressed for hours isn't re-prompting
|
||||||
|
// the LLM every tick it stays blocked (EnqueueDigestEntry's rule+body dedupe
|
||||||
|
// makes repeat calls here harmless, but skipping the phrase call entirely
|
||||||
|
// when a pending entry already exists avoids the LLM round-trip too).
|
||||||
|
func (t *tickLoop) enqueueSuppressedDigest(ctx context.Context, trace *loop.TickTrace, state loop.State, now time.Time) {
|
||||||
|
if trace == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
for _, tr := range trace.RuleTraces {
|
||||||
|
if !tr.PredicateResult || tr.GateResult {
|
||||||
|
continue // didn't want to fire, or wasn't suppressed
|
||||||
|
}
|
||||||
|
if !loop.DigestEligible(tr.Severity, tr.GateBlockedBy) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
rule := loop.Rule{Name: tr.RuleName, Severity: tr.Severity}
|
||||||
|
cand := loop.Candidate{Rule: rule, Severity: tr.Severity, State: state}
|
||||||
|
pn, err := t.phraser.PhraseNudge(ctx, cand)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: phrase digest candidate %s: %v", tr.RuleName, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
expires := now.Add(digestExpiry)
|
||||||
|
if _, deduped, err := t.store.EnqueueDigestEntry(ctx, tr.RuleName, int(tr.Severity), pn.Body, now, expires); err != nil {
|
||||||
|
log.Printf("tick: enqueue digest entry %s: %v", tr.RuleName, err)
|
||||||
|
} else if deduped {
|
||||||
|
// same suppressed nudge already pending — nothing new to say.
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// expireStaleDigest sweeps entries past their expiry once per tick — cheap
|
||||||
|
// bookkeeping, mirrors ReconcileStaleDeliveryAttempts's shape.
|
||||||
|
func (t *tickLoop) expireStaleDigest(ctx context.Context, now time.Time) {
|
||||||
|
n, err := t.store.ExpireStaleDigestEntries(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: expire stale digest entries: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if n > 0 {
|
||||||
|
log.Printf("tick: expired %d stale digest entr(y/ies) unspoken", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// maybeDrainDigest speaks the pending digest bundle once the gate's
|
||||||
|
// suppression reasons have actually cleared — quiet hours over, back from
|
||||||
|
// away, out of the meeting. Draining while still suppressed would just be a
|
||||||
|
// second way to nag through quiet hours; the bundle waits for the same "is
|
||||||
|
// it allowed right now" condition a live nudge already waits for.
|
||||||
|
func (t *tickLoop) maybeDrainDigest(ctx context.Context, state loop.State, now time.Time) {
|
||||||
|
if state.QuietHours || state.CalendarBusy || state.Presence == store.Away {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
entries, err := t.store.PendingDigestEntries(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: pending digest entries: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if len(entries) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
spoken := entries
|
||||||
|
extra := 0
|
||||||
|
if len(spoken) > maxDigestSpokenItems {
|
||||||
|
spoken = entries[:maxDigestSpokenItems]
|
||||||
|
extra = len(entries) - maxDigestSpokenItems
|
||||||
|
}
|
||||||
|
var b strings.Builder
|
||||||
|
maxSev := 0
|
||||||
|
for i, e := range spoken {
|
||||||
|
if i > 0 {
|
||||||
|
b.WriteString(" · ")
|
||||||
|
}
|
||||||
|
b.WriteString(e.Body)
|
||||||
|
if e.Severity > maxSev {
|
||||||
|
maxSev = e.Severity
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if extra > 0 {
|
||||||
|
fmt.Fprintf(&b, " · и ещё %d", extra)
|
||||||
|
}
|
||||||
|
body := b.String()
|
||||||
|
summary := fmt.Sprintf("%d отложенных уведомлений", len(entries))
|
||||||
|
|
||||||
|
cand := loop.Candidate{
|
||||||
|
Rule: loop.Rule{Name: "digest", Severity: loop.Severity(maxSev)},
|
||||||
|
Severity: loop.Severity(maxSev),
|
||||||
|
State: state,
|
||||||
|
}
|
||||||
|
pn := delivery.PhrasedNudge{Candidate: cand, Body: body, Summary: summary}
|
||||||
|
t.cachePhrase(pn)
|
||||||
|
if _, err := t.dispatcher.DispatchNudge(ctx, pn, now); err != nil {
|
||||||
|
log.Printf("tick: dispatch digest bundle: %v", err)
|
||||||
|
return // leave entries pending; retried next tick
|
||||||
|
}
|
||||||
|
ids := make([]int64, len(entries))
|
||||||
|
for i, e := range entries {
|
||||||
|
ids[i] = e.ID
|
||||||
|
}
|
||||||
|
if err := t.store.DrainDigestEntries(ctx, ids, now); err != nil {
|
||||||
|
log.Printf("tick: drain digest entries: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// routinesFromConfig maps the config's routine blocks to the engine type.
|
// routinesFromConfig maps the config's routine blocks to the engine type.
|
||||||
// Validation (cron parses, name/body present, severity defaulted) already ran
|
// Validation (cron parses, name/body present, severity defaulted) already ran
|
||||||
// in config.Load, so this is a pure field copy.
|
// in config.Load, so this is a pure field copy.
|
||||||
@@ -377,6 +640,64 @@ func (t *tickLoop) fireRoutines(ctx context.Context, now time.Time, state loop.S
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// fireAcceptedRoutines nudges about the routines the user accepted, once per
|
||||||
|
// interval (Vikunja #366). Accepting used to create a single reminder, so a
|
||||||
|
// non-weekly routine fired once and went quiet forever; the schedule lives in
|
||||||
|
// the proposed_routines row now and the loop re-reads it every tick.
|
||||||
|
//
|
||||||
|
// A routine is a care-class nudge and goes through the restraint gate like any
|
||||||
|
// other: quiet hours, away presence and snooze all suppress it. Reminders bypass
|
||||||
|
// that gate; routines must not. A suppressed nudge is NOT marked fired, so it
|
||||||
|
// goes out on the next tick that the gate allows — one nudge, held, not dropped
|
||||||
|
// and not repeated.
|
||||||
|
//
|
||||||
|
// The body is literal text built from the detected action and object, not
|
||||||
|
// LLM-phrased, so a routine can't hallucinate. It nudges; it never acts.
|
||||||
|
func (t *tickLoop) fireAcceptedRoutines(ctx context.Context, now time.Time, state loop.State) {
|
||||||
|
rows, err := t.store.ListAcceptedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: list accepted routines: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
accepted := make([]routine.Accepted, 0, len(rows))
|
||||||
|
for _, r := range rows {
|
||||||
|
if r.AcceptedTs == nil {
|
||||||
|
continue // accepted before the schedule column existed — no clock to start from.
|
||||||
|
}
|
||||||
|
accepted = append(accepted, routine.Accepted{
|
||||||
|
ID: r.ID,
|
||||||
|
Name: r.Action + " " + r.Object,
|
||||||
|
IntervalDays: r.IntervalDays,
|
||||||
|
Accepted: *r.AcceptedTs,
|
||||||
|
LastFired: r.LastFiredTs,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, a := range routine.DueAccepted(accepted, now) {
|
||||||
|
rule := loop.Rule{Name: "routine:" + a.Name, Severity: loop.Sev1}
|
||||||
|
if !loop.Gate(state, rule) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
body := "пора: " + a.Name
|
||||||
|
pn := delivery.PhrasedNudge{
|
||||||
|
Candidate: loop.Candidate{Rule: rule, Severity: rule.Severity, State: state},
|
||||||
|
Body: body,
|
||||||
|
Summary: body,
|
||||||
|
}
|
||||||
|
sent, err := t.dispatcher.DispatchNudge(ctx, pn, now)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("tick: dispatch accepted routine %d: %v", a.ID, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if len(sent) == 0 {
|
||||||
|
continue // routing dropped it — leave it due.
|
||||||
|
}
|
||||||
|
if err := t.store.MarkRoutineFired(ctx, a.ID, now); err != nil {
|
||||||
|
log.Printf("tick: mark routine %d fired: %v", a.ID, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// fireMorningRoutines checks each configured checklist against today's facts
|
// fireMorningRoutines checks each configured checklist against today's facts
|
||||||
// and dispatches a nag listing exactly what's still missing, at most once per
|
// and dispatches a nag listing exactly what's still missing, at most once per
|
||||||
// routine per calendar day. Fact reads happen here (not in loop.Gatherer)
|
// routine per calendar day. Fact reads happen here (not in loop.Gatherer)
|
||||||
|
|||||||
+111
-2
@@ -46,7 +46,7 @@ func newTestTickLoop(t *testing.T, st *store.Store, sink delivery.Sink, digestCf
|
|||||||
Nudges: st,
|
Nudges: st,
|
||||||
Reminders: st,
|
Reminders: st,
|
||||||
})
|
})
|
||||||
return newTickLoop(st, g, d, phraser.NewStub(), rules, time.Second, 5*time.Minute, 0, digestCfg, nil, nil)
|
return newTickLoop(st, g, d, phraser.NewStub(), rules, time.Second, 5*time.Minute, 0, digestCfg, nil, nil, nil)
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestTickFiresRoutineWhenScheduleCrosses(t *testing.T) {
|
func TestTickFiresRoutineWhenScheduleCrosses(t *testing.T) {
|
||||||
@@ -63,7 +63,7 @@ func TestTickFiresRoutineWhenScheduleCrosses(t *testing.T) {
|
|||||||
sink := &fakeSink{}
|
sink := &fakeSink{}
|
||||||
d := delivery.NewDispatcher(delivery.Config{Voice: sink, Ntfy: sink, Telegram: sink, Nudges: st, Reminders: st})
|
d := delivery.NewDispatcher(delivery.Config{Voice: sink, Ntfy: sink, Telegram: sink, Nudges: st, Reminders: st})
|
||||||
rs := []routine.Routine{{Name: "morning", Cron: "0 12 * * *", Body: "полдень, время воды", Severity: 1}}
|
rs := []routine.Routine{{Name: "morning", Cron: "0 12 * * *", Body: "полдень, время воды", Severity: 1}}
|
||||||
tl := newTickLoop(st, g, d, phraser.NewStub(), rules, time.Second, 5*time.Minute, 0, nil, rs, nil)
|
tl := newTickLoop(st, g, d, phraser.NewStub(), rules, time.Second, 5*time.Minute, 0, nil, rs, nil, nil)
|
||||||
|
|
||||||
// first tick: seeds, does not fire the routine.
|
// first tick: seeds, does not fire the routine.
|
||||||
tl.tick(ctx, now)
|
tl.tick(ctx, now)
|
||||||
@@ -96,6 +96,115 @@ func TestTickFiresRoutineWhenScheduleCrosses(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestTickFiresAcceptedRoutineEveryInterval — Vikunja #366. An accepted routine
|
||||||
|
// with a 3-day interval must nudge every 3 days, not once. It also must not
|
||||||
|
// replay the occurrences it slept through: after a 30-day gap it nudges once.
|
||||||
|
func TestTickFiresAcceptedRoutineEveryInterval(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
accepted := refNow()
|
||||||
|
|
||||||
|
id, err := st.CreateProposedRoutine(ctx, "полить", "цветы", 3.0, accepted)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CreateProposedRoutine: %v", err)
|
||||||
|
}
|
||||||
|
if err := st.AcceptProposedRoutine(ctx, id, accepted); err != nil {
|
||||||
|
t.Fatalf("AcceptProposedRoutine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
const rule = "routine:полить цветы"
|
||||||
|
|
||||||
|
// Same day as the accept: not due yet.
|
||||||
|
markPresent(t, st, ctx, accepted)
|
||||||
|
tl.tick(ctx, accepted.Add(time.Hour))
|
||||||
|
if n := countSends(sink, rule); n != 0 {
|
||||||
|
t.Fatalf("routine fired %d times before its first interval passed, want 0", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Three days later: the first nudge.
|
||||||
|
first := accepted.Add(3 * 24 * time.Hour)
|
||||||
|
markPresent(t, st, ctx, first)
|
||||||
|
tl.tick(ctx, first)
|
||||||
|
if n := countSends(sink, rule); n != 1 {
|
||||||
|
t.Fatalf("first interval: sends = %d, want 1", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Next day: still inside the interval, silent.
|
||||||
|
sink.sends = nil
|
||||||
|
markPresent(t, st, ctx, first.Add(24*time.Hour))
|
||||||
|
tl.tick(ctx, first.Add(24*time.Hour))
|
||||||
|
if n := countSends(sink, rule); n != 0 {
|
||||||
|
t.Fatalf("mid-interval: sends = %d, want 0", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Three days after the first nudge: it fires again. This is the bug —
|
||||||
|
// a one-shot reminder would never come back.
|
||||||
|
second := first.Add(3 * 24 * time.Hour)
|
||||||
|
markPresent(t, st, ctx, second)
|
||||||
|
tl.tick(ctx, second)
|
||||||
|
if n := countSends(sink, rule); n != 1 {
|
||||||
|
t.Fatalf("second interval: sends = %d, want 1 (a routine repeats)", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// A long silence must not turn into a backlog of missed nudges.
|
||||||
|
sink.sends = nil
|
||||||
|
late := second.Add(30 * 24 * time.Hour)
|
||||||
|
markPresent(t, st, ctx, late)
|
||||||
|
tl.tick(ctx, late)
|
||||||
|
if n := countSends(sink, rule); n != 1 {
|
||||||
|
t.Fatalf("after a 30-day gap: sends = %d, want exactly 1 (no backlog)", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestTickAcceptedRoutineRespectsQuietHours — routines are not reminders: they
|
||||||
|
// do not inherit the reminder gate bypass. Away presence drops a care-class
|
||||||
|
// nudge, and the routine stays due so it nudges once the user is back.
|
||||||
|
func TestTickAcceptedRoutineRespectsGate(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
accepted := refNow()
|
||||||
|
|
||||||
|
id, err := st.CreateProposedRoutine(ctx, "полить", "цветы", 3.0, accepted)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("CreateProposedRoutine: %v", err)
|
||||||
|
}
|
||||||
|
if err := st.AcceptProposedRoutine(ctx, id, accepted); err != nil {
|
||||||
|
t.Fatalf("AcceptProposedRoutine: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sink := &fakeSink{}
|
||||||
|
tl := newTestTickLoop(t, st, sink, nil)
|
||||||
|
const rule = "routine:полить цветы"
|
||||||
|
|
||||||
|
// No presence probes at all ⇒ away ⇒ the care gate blocks the nudge.
|
||||||
|
due := accepted.Add(3 * 24 * time.Hour)
|
||||||
|
tl.tick(ctx, due)
|
||||||
|
if n := countSends(sink, rule); n != 0 {
|
||||||
|
t.Fatalf("away: sends = %d, want 0 (routine must not bypass the gate)", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Back at the desk a minute later: the nudge that was held now goes out.
|
||||||
|
back := due.Add(time.Minute)
|
||||||
|
markPresent(t, st, ctx, back)
|
||||||
|
tl.tick(ctx, back)
|
||||||
|
if n := countSends(sink, rule); n != 1 {
|
||||||
|
t.Fatalf("present again: sends = %d, want 1", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// countSends counts captured sends for one rule name.
|
||||||
|
func countSends(sink *fakeSink, rule string) int {
|
||||||
|
n := 0
|
||||||
|
for _, s := range sink.sends {
|
||||||
|
if s.RuleName == rule {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
|
||||||
// refNow — fixed tick time so presence decay + since durations are deterministic.
|
// refNow — fixed tick time so presence decay + since durations are deterministic.
|
||||||
func refNow() time.Time { return time.Date(2026, 6, 30, 12, 0, 0, 0, time.UTC) }
|
func refNow() time.Time { return time.Date(2026, 6, 30, 12, 0, 0, 0, time.UTC) }
|
||||||
|
|
||||||
|
|||||||
+100
-1450
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,434 @@
|
|||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bufio"
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/config"
|
||||||
|
"github.com/kami/maven/internal/delivery"
|
||||||
|
"github.com/kami/maven/internal/delivery/voicesink"
|
||||||
|
"github.com/kami/maven/internal/dialogue"
|
||||||
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
"github.com/kami/maven/internal/memory"
|
||||||
|
"github.com/kami/maven/internal/phraser"
|
||||||
|
"github.com/kami/maven/internal/router"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
"github.com/kami/maven/internal/stt"
|
||||||
|
"github.com/kami/maven/internal/tool"
|
||||||
|
"github.com/kami/maven/internal/tts"
|
||||||
|
"github.com/kami/maven/internal/voice"
|
||||||
|
"github.com/kami/maven/internal/weather"
|
||||||
|
"github.com/kami/maven/internal/worker"
|
||||||
|
)
|
||||||
|
|
||||||
|
// voiceWiring — everything the daemon needs to run the audio path. Held by
|
||||||
|
// cmd/mavend/main.go alongside the other wirings; closed on shutdown.
|
||||||
|
type voiceWiring struct {
|
||||||
|
server *voice.Server
|
||||||
|
sessions *voice.Sessions
|
||||||
|
voiceSink delivery.Sink
|
||||||
|
embedder router.Embedder
|
||||||
|
handler *reactiveHandler // the reactive handler for IPC Chat
|
||||||
|
// worker clients (set when configured as Remote): closed on shutdown so
|
||||||
|
// mavsttd / mavttsd don't keep a stale conn into a restarting daemon.
|
||||||
|
sttClient *worker.Client
|
||||||
|
ttsClient *worker.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// close releases the listener + worker conns. Safe to call on nil (when
|
||||||
|
// voice is not wired — wireVoice returns nil,nil).
|
||||||
|
func (w *voiceWiring) close() {
|
||||||
|
if w == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if w.embedder != nil {
|
||||||
|
_ = w.embedder.Close()
|
||||||
|
}
|
||||||
|
if w.server != nil {
|
||||||
|
_ = w.server.Close()
|
||||||
|
}
|
||||||
|
if w.sttClient != nil {
|
||||||
|
_ = w.sttClient.Close()
|
||||||
|
}
|
||||||
|
if w.ttsClient != nil {
|
||||||
|
_ = w.ttsClient.Close()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// wireVoice builds the audio path from cfg + a CoreAPI + a router. Returns
|
||||||
|
// nil wiring + nil error when voice isn't enabled (the caller's voice sink
|
||||||
|
// stays nil; the dispatcher's ChannelVoice routing drops silently).
|
||||||
|
//
|
||||||
|
// When voice is enabled, MUST wire a voicesink into the dispatcher's Voice
|
||||||
|
// slot using w.sessions (the caller does that — see main.go).
|
||||||
|
func wireVoice(cfg *config.Config, coreAPI ipc.CoreAPI, phr phraser.Phraser, memStore memory.Store, dataStore *store.Store, eco *ecosystemWiring) (*voiceWiring, error) {
|
||||||
|
if cfg.Voice == nil || !cfg.Voice.Enabled {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
w := &voiceWiring{}
|
||||||
|
|
||||||
|
// ----- stt (Stub in-process OR Remote via worker socket) -----
|
||||||
|
var transcriber stt.Transcriber
|
||||||
|
if cfg.Voice.Stt != nil && cfg.Voice.Stt.Socket != "" {
|
||||||
|
c := worker.Dial(cfg.Voice.Stt.Socket)
|
||||||
|
w.sttClient = c
|
||||||
|
lang := cfg.Voice.Stt.Lang
|
||||||
|
if lang == "" {
|
||||||
|
lang = cfg.Voice.Lang
|
||||||
|
}
|
||||||
|
transcriber = stt.NewRemote(c, lang)
|
||||||
|
} else {
|
||||||
|
transcriber = stt.NewStub()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- tts (Stub in-process OR Remote) -----
|
||||||
|
var synthesizer tts.Synthesizer
|
||||||
|
if cfg.Voice.Tts != nil && cfg.Voice.Tts.Socket != "" {
|
||||||
|
c := worker.Dial(cfg.Voice.Tts.Socket)
|
||||||
|
w.ttsClient = c
|
||||||
|
lang := cfg.Voice.Tts.Lang
|
||||||
|
if lang == "" {
|
||||||
|
lang = cfg.Voice.Lang
|
||||||
|
}
|
||||||
|
synthesizer = tts.NewRemote(c, lang, cfg.Voice.Tts.Voice)
|
||||||
|
} else {
|
||||||
|
synthesizer = tts.NewStub()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- router: embedder (ONNX when configured, floor HashEmbedder otherwise) -----
|
||||||
|
var emb router.Embedder
|
||||||
|
if cfg.Voice.Embedder != nil {
|
||||||
|
onnx, err := router.NewONNXEmbedder(
|
||||||
|
cfg.Voice.Embedder.ModelPath,
|
||||||
|
cfg.Voice.Embedder.TokenizerPath,
|
||||||
|
cfg.Voice.Embedder.LibPath,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
w.close()
|
||||||
|
return nil, fmt.Errorf("embedder: %w", err)
|
||||||
|
}
|
||||||
|
log.Printf("voice: onnx embedder loaded (%d dim)", onnx.Dim())
|
||||||
|
emb = onnx
|
||||||
|
} else {
|
||||||
|
log.Printf("voice: embedder not configured, using HashEmbedder floor")
|
||||||
|
emb = router.NewHashEmbedder(1024)
|
||||||
|
}
|
||||||
|
w.embedder = emb
|
||||||
|
checkStoredEmbedder(dataStore, emb)
|
||||||
|
|
||||||
|
// ----- tool executor (the enabled act allowlist, store-backed) -----
|
||||||
|
// Config tools are the declarative bootstrap: seed them into the store as
|
||||||
|
// enabled (editing mavend.json IS the human enable act). Ad-hoc tools are
|
||||||
|
// enabled later through the authed mavweb surface. The executor + matcher
|
||||||
|
// both read the store live, so a newly-enabled tool is runnable without a
|
||||||
|
// daemon restart.
|
||||||
|
seedTools(coreAPI, cfg.Voice.Tools)
|
||||||
|
exec := tool.NewExecutor(coreAPI, time.Duration(cfg.Voice.ToolTimeout))
|
||||||
|
matcher := tool.NewMatcher(coreAPI)
|
||||||
|
|
||||||
|
// ----- weather provider (Open-Meteo when configured, Stub otherwise) -----
|
||||||
|
var weatherProvider weather.Provider
|
||||||
|
var weatherLocation string
|
||||||
|
if cfg.Voice.Weather != nil && cfg.Voice.Weather.Provider == "open-meteo" {
|
||||||
|
weatherProvider = weather.NewOpenMeteoProvider()
|
||||||
|
weatherLocation = cfg.Voice.Weather.DefaultLocation
|
||||||
|
log.Printf("voice: weather provider: open-meteo (default location: %s)", cfg.Voice.Weather.DefaultLocation)
|
||||||
|
} else {
|
||||||
|
weatherProvider = weather.NewStubProvider()
|
||||||
|
log.Printf("voice: weather provider: stub (not configured)")
|
||||||
|
}
|
||||||
|
|
||||||
|
// The replier uses the same llama-server as the phraser.
|
||||||
|
var llmClient *llm.Client
|
||||||
|
if lp, ok := phr.(*phraser.LLMPhraser); ok {
|
||||||
|
llmClient = llm.New(lp.BaseURL(), 60*time.Second)
|
||||||
|
}
|
||||||
|
// ----- router (the cascade; floor examples seed the classifier) -----
|
||||||
|
// The act matcher's allowlist is exactly the enabled tool names — the
|
||||||
|
// router only matches acts the executor can run (one source of truth).
|
||||||
|
threshold := cfg.Voice.RouterThreshold
|
||||||
|
if threshold <= 0 {
|
||||||
|
threshold = config.DefaultRouterThreshold
|
||||||
|
}
|
||||||
|
// The resident model routes by default: 63.2% of held-out intents right
|
||||||
|
// against the classifier's 50.0%, at about 1s a turn instead of 30ms (see
|
||||||
|
// config.VoiceConfig.LLMRouter). The classifier always stays wired as the
|
||||||
|
// fallback, so a model error never breaks a turn.
|
||||||
|
rtr := buildRouter(emb, matcher, threshold, pickLLMRouter(cfg.Voice.UseLLMRouter(), llmClient))
|
||||||
|
|
||||||
|
// ----- sessions registry (shared with voicesink) -----
|
||||||
|
sessions := voice.NewSessions()
|
||||||
|
w.sessions = sessions
|
||||||
|
|
||||||
|
// ----- voice sink (proactive nudges: dispatcher → voicesink → tts → push to client) -----
|
||||||
|
w.voiceSink = voicesink.New(synthesizer, sessions)
|
||||||
|
|
||||||
|
// ----- memory (long-term vector storage) -----
|
||||||
|
// Persistent (store-backed, survives restarts) when the daemon passes one;
|
||||||
|
// falls back to the in-memory floor otherwise (tests / no-store paths).
|
||||||
|
if memStore == nil {
|
||||||
|
memStore = memory.NewInMemoryStore()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- dialogue (multi-turn slot carry-over; 2-min follow-up window) -----
|
||||||
|
// Store-backed when the daemon passes a store, so a restart mid-conversation
|
||||||
|
// keeps the thread (Vikunja #363). Sessions past their TTL are dropped on
|
||||||
|
// load, never revived. Clarify's parked question stays in memory only.
|
||||||
|
var dialogueSessions *dialogue.SessionStore
|
||||||
|
if dataStore != nil {
|
||||||
|
dialogueSessions = dialogue.NewPersistentSessionStore(2*time.Minute, dataStore)
|
||||||
|
if err := dialogueSessions.Load(context.Background(), time.Now()); err != nil {
|
||||||
|
log.Printf("dialogue: load saved sessions: %v", err)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
dialogueSessions = dialogue.NewSessionStore(2 * time.Minute)
|
||||||
|
}
|
||||||
|
clarifyStore := dialogue.NewClarifyStore(clarifyTTL)
|
||||||
|
timeParser := router.NewPythonDateParser()
|
||||||
|
|
||||||
|
// ----- replier (LLM-backed when the engine is on, Stub floor otherwise) -----
|
||||||
|
replier := voice.Replier(voice.NewStubReplier())
|
||||||
|
if llmClient != nil {
|
||||||
|
replier = newLLMReplier(llmClient, contextBlockFn(cfg, time.Now))
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- the handler (the reactive path; closes over stt / tts / router / coreAPI / memory) -----
|
||||||
|
h := &reactiveHandler{
|
||||||
|
stt: transcriber,
|
||||||
|
tts: synthesizer,
|
||||||
|
router: rtr,
|
||||||
|
embedder: emb,
|
||||||
|
api: coreAPI,
|
||||||
|
tools: exec,
|
||||||
|
matcher: matcher,
|
||||||
|
replier: replier,
|
||||||
|
phraser: phr,
|
||||||
|
now: time.Now,
|
||||||
|
weatherProvider: weatherProvider,
|
||||||
|
weatherLocation: weatherLocation,
|
||||||
|
memStore: memStore,
|
||||||
|
dataStore: dataStore,
|
||||||
|
dialogueSessions: dialogueSessions,
|
||||||
|
clarifyStore: clarifyStore,
|
||||||
|
// 0 here (unset config) ⇒ the dialogue default.
|
||||||
|
clarifyMaxAttempts: cfg.Voice.ClarifyMaxAttempts,
|
||||||
|
extractor: router.Extractor{Time: timeParser, Acts: matcher, Facts: router.DefaultFactParser{}},
|
||||||
|
queryMinScore: cfg.Voice.QueryMinScore,
|
||||||
|
queryMinMargin: cfg.Voice.QueryMinMargin,
|
||||||
|
timeParser: timeParser,
|
||||||
|
ecosystem: eco,
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- the server (TCP listener) -----
|
||||||
|
srv := voice.NewServer(cfg.Voice.Bind, h, sessions)
|
||||||
|
if err := srv.Listen(); err != nil {
|
||||||
|
w.close()
|
||||||
|
return nil, fmt.Errorf("voice listen: %w", err)
|
||||||
|
}
|
||||||
|
w.server = srv
|
||||||
|
w.handler = h
|
||||||
|
|
||||||
|
return w, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// pickLLMRouter returns the LLM router when the operator asked for it and there
|
||||||
|
// is a llama-server to talk to, and nil otherwise. nil is safe: the cascade then
|
||||||
|
// routes with the classifier, so an unusable setting costs accuracy, not turns.
|
||||||
|
func pickLLMRouter(enabled bool, c *llm.Client) *router.LLMRouter {
|
||||||
|
if !enabled {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if c == nil {
|
||||||
|
log.Printf("voice: voice.llm_router is on but there is no llama-server to route with (the phraser is not an LLM phraser) — using the classifier instead")
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
log.Printf("voice: LLM router enabled")
|
||||||
|
return router.NewLLMRouter(c)
|
||||||
|
}
|
||||||
|
|
||||||
|
// buildRouter constructs the reactive-path router with the given embedder
|
||||||
|
// and confidence threshold.
|
||||||
|
// - stage-0 grammars from DefaultActMatcher whose fn allowlist is exactly
|
||||||
|
// the enabled tool names (actFns) — the router only matches acts the
|
||||||
|
// executor can run. Empty ⇒ every act refuses at the matcher.
|
||||||
|
// - The embedder is provided by wireVoice: HashEmbedder (floor) when no
|
||||||
|
// embedder config is present, or the ONNX multilingual model when
|
||||||
|
// configured — same interface, one constructor change.
|
||||||
|
// - 6 bootstrap examples covering the 5 intents + one compound-capture
|
||||||
|
// placeholder. Spec calls for ~10 per intent at production; this is the
|
||||||
|
// bootstrapping floor swapped by tuning the seed set later.
|
||||||
|
// - Threshold is from voice.router_threshold config (default 0.55).
|
||||||
|
func buildRouter(emb router.Embedder, acts router.ActMatcher, threshold float64, llmR *router.LLMRouter) *router.Router {
|
||||||
|
cls := router.NewClassifier(emb)
|
||||||
|
seedClassifier(cls)
|
||||||
|
grammars := router.DefaultGrammars(acts)
|
||||||
|
grammars = append(grammars, router.SystemTimeDateGrammars()...)
|
||||||
|
grammars = append(grammars, router.ReminderGrammar())
|
||||||
|
return router.New(router.Config{
|
||||||
|
Grammars: grammars,
|
||||||
|
Classifier: cls,
|
||||||
|
Extractor: router.Extractor{
|
||||||
|
Time: router.NewPythonDateParser(),
|
||||||
|
Acts: acts,
|
||||||
|
Facts: router.DefaultFactParser{},
|
||||||
|
},
|
||||||
|
Threshold: threshold,
|
||||||
|
LLM: llmR,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedDir is the directory containing intent seed files. Each file is named
|
||||||
|
// <intent>.txt and contains one training example per line (blank lines and
|
||||||
|
// lines starting with # are ignored). Relative to the working directory.
|
||||||
|
const seedDir = "models/seeds"
|
||||||
|
|
||||||
|
// seedClassifier floors the embedded examples so the cold-boot path
|
||||||
|
// doesn't return ErrNoIntents. Loads examples from seedDir — one file per
|
||||||
|
// intent (act.txt, reminder.txt, fact.txt, note.txt, query.txt). When the
|
||||||
|
// classifier can't decide it falls through to Clarify — the last-resort
|
||||||
|
// path asks the user to rephrase rather than guessing wrong.
|
||||||
|
func seedClassifier(c *router.Classifier) {
|
||||||
|
intents := []router.Intent{
|
||||||
|
router.IntentAct,
|
||||||
|
router.IntentReminder,
|
||||||
|
router.IntentFact,
|
||||||
|
router.IntentNote,
|
||||||
|
router.IntentQuery,
|
||||||
|
router.IntentChat,
|
||||||
|
router.IntentSystem,
|
||||||
|
}
|
||||||
|
total := 0
|
||||||
|
for _, intent := range intents {
|
||||||
|
n, err := loadSeedFile(c, intent)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: seed %s: %v", intent, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
total += n
|
||||||
|
}
|
||||||
|
log.Printf("voice: loaded %d seed examples from %s", total, seedDir)
|
||||||
|
}
|
||||||
|
|
||||||
|
func loadSeedFile(c *router.Classifier, intent router.Intent) (int, error) {
|
||||||
|
path := filepath.Join(seedDir, string(intent)+".txt")
|
||||||
|
f, err := os.Open(path)
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("open %s: %w", path, err)
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
|
||||||
|
var count int
|
||||||
|
sc := bufio.NewScanner(f)
|
||||||
|
for sc.Scan() {
|
||||||
|
line := strings.TrimSpace(sc.Text())
|
||||||
|
if line == "" || strings.HasPrefix(line, "#") {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := c.AddExample(context.Background(), intent, line); err != nil {
|
||||||
|
log.Printf("voice: seed %s: skipping %q: %v", intent, line, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
count++
|
||||||
|
}
|
||||||
|
if err := sc.Err(); err != nil {
|
||||||
|
return count, fmt.Errorf("scan %s: %w", path, err)
|
||||||
|
}
|
||||||
|
return count, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// seedTools upserts the config-declared tools into the store as enabled. Editing
|
||||||
|
// mavend.json is a human act, so a config tool is enabled by definition; this
|
||||||
|
// makes the declarative config the reproducible bootstrap while the store stays
|
||||||
|
// the single runtime source of truth (mavweb enables ad-hoc ones on top).
|
||||||
|
func seedTools(api ipc.CoreAPI, tools []config.ToolConfig) {
|
||||||
|
ctx := context.Background()
|
||||||
|
now := time.Now()
|
||||||
|
n := 0
|
||||||
|
for _, tc := range tools {
|
||||||
|
if tc.Name == "" || len(tc.Cmd) == 0 {
|
||||||
|
log.Printf("voice: skipping malformed tool config %+v", tc)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := api.EnableTool(ctx, tc.Name, tc.Cmd, tc.Destructive, tc.Scope, now); err != nil {
|
||||||
|
log.Printf("voice: seed tool %q: %v", tc.Name, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
log.Printf("voice: seeded %d act tools from config", n)
|
||||||
|
}
|
||||||
|
|
||||||
|
// reembedOnStart is the -reembed flag (set in run()). Opt-in on purpose: see
|
||||||
|
// runReembed.
|
||||||
|
var reembedOnStart bool
|
||||||
|
|
||||||
|
// checkStoredEmbedder compares the embedder we just loaded with the one that
|
||||||
|
// wrote the vectors already in the DB (Vikunja #378).
|
||||||
|
//
|
||||||
|
// The two models we have both make 384-dim vectors, so a size check catches
|
||||||
|
// nothing: after a swap, recall silently compares vectors from different
|
||||||
|
// spaces and the scores are noise. So we say it out loud. Recall itself is not
|
||||||
|
// changed here — the fix is `mavend -reembed`.
|
||||||
|
func checkStoredEmbedder(dataStore *store.Store, emb router.Embedder) {
|
||||||
|
if dataStore == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
current := router.EmbedderID(emb)
|
||||||
|
if reembedOnStart {
|
||||||
|
runReembed(dataStore, emb, current)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
stored, mismatch, err := dataStore.CheckEmbedder(context.Background(), current)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: embedder marker check failed: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if mismatch {
|
||||||
|
log.Printf("voice: WARNING embedder MISMATCH — stored vectors were written by %q but the configured embedder is %q; recall scores are noise until the notes and facts are re-embedded — run `mavend -reembed` once (Vikunja #378)", stored, current)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
log.Printf("voice: embedder marker ok (%s)", current)
|
||||||
|
}
|
||||||
|
|
||||||
|
// runReembed is the one-shot backfill behind -reembed.
|
||||||
|
//
|
||||||
|
// Why a flag and not automatic on mismatch: the embedder is ONNX on the
|
||||||
|
// laptop's CPU, so a few thousand notes is minutes of work. Doing that silently
|
||||||
|
// inside a normal start would look like the daemon hanging on boot. So the user
|
||||||
|
// runs it once, deliberately, after an embedder swap; the mismatch warning
|
||||||
|
// above tells them to. It re-embeds, logs what it did, and then the daemon
|
||||||
|
// carries on serving as usual — no separate binary, no second start needed.
|
||||||
|
func runReembed(dataStore *store.Store, emb router.Embedder, current string) {
|
||||||
|
log.Printf("voice: re-embedding stored notes and facts with %s — this can take a few minutes, do not interrupt", current)
|
||||||
|
res, err := dataStore.ReembedAll(context.Background(), current,
|
||||||
|
// EmbedPassage, not EmbedQuery: these are stored texts being searched
|
||||||
|
// FOR, which is the side they were written with.
|
||||||
|
func(ctx context.Context, text string) ([]float32, error) {
|
||||||
|
return router.EmbedPassage(ctx, emb, text)
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("voice: re-embed FAILED, nothing was changed and no marker was written — safe to run again: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if res.Skipped {
|
||||||
|
log.Printf("voice: re-embed skipped — the stored vectors were already written by %s", current)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
log.Printf("voice: re-embed done — %d notes in the notes table, %d notes and %d facts in the memory index, took %s; stored vectors now belong to %s",
|
||||||
|
res.Notes, res.MemNotes, res.Facts, res.Took.Round(time.Second), current)
|
||||||
|
|
||||||
|
// A row with no text cannot be re-embedded, so its vector is still the old
|
||||||
|
// model's noise while the marker now says everything is current. Both write
|
||||||
|
// paths always store the text, so this should be zero — say it loudly
|
||||||
|
// rather than bury it in the line above if it ever isn't.
|
||||||
|
if res.NoText > 0 {
|
||||||
|
log.Printf("voice: WARNING %d stored rows had no text, so their vectors could not be re-embedded and are still noise; they will never match anything useful (Vikunja #378)", res.NoText)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,50 @@
|
|||||||
|
// Package main — weatherq.go holds the weather-query keyword helpers: does
|
||||||
|
// this utterance ask about weather at all, and which city (if any) did it
|
||||||
|
// name. Both are plain substring/lookup matching, not NLU — extend this file
|
||||||
|
// rather than voice.go for anything in that shape.
|
||||||
|
package main
|
||||||
|
|
||||||
|
import "strings"
|
||||||
|
|
||||||
|
// isWeatherQuery returns true if the utterance is about weather.
|
||||||
|
func isWeatherQuery(u string) bool {
|
||||||
|
lower := strings.ToLower(u)
|
||||||
|
return strings.Contains(lower, "погод") ||
|
||||||
|
strings.Contains(lower, "градус") ||
|
||||||
|
strings.Contains(lower, "температур") ||
|
||||||
|
strings.Contains(lower, "дожд") ||
|
||||||
|
strings.Contains(lower, "холод") ||
|
||||||
|
strings.Contains(lower, "тепл") ||
|
||||||
|
strings.Contains(lower, "weather") ||
|
||||||
|
strings.Contains(lower, "temperature")
|
||||||
|
}
|
||||||
|
|
||||||
|
// extractWeatherLocation parses a location from the utterance, or falls back
|
||||||
|
// to the configured default. Very basic: just checks for known city names.
|
||||||
|
func extractWeatherLocation(u, defaultLoc string) string {
|
||||||
|
lower := strings.ToLower(u)
|
||||||
|
cities := map[string]string{
|
||||||
|
"москв": "Moscow",
|
||||||
|
"moscow": "Moscow",
|
||||||
|
"питер": "Saint Petersburg",
|
||||||
|
"spb": "Saint Petersburg",
|
||||||
|
"петербур": "Saint Petersburg",
|
||||||
|
"лондон": "London",
|
||||||
|
"london": "London",
|
||||||
|
"париж": "Paris",
|
||||||
|
"paris": "Paris",
|
||||||
|
"берлин": "Berlin",
|
||||||
|
"berlin": "Berlin",
|
||||||
|
"нью-йорк": "New York",
|
||||||
|
"new york": "New York",
|
||||||
|
}
|
||||||
|
for substr, name := range cities {
|
||||||
|
if strings.Contains(lower, substr) {
|
||||||
|
return name
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if defaultLoc != "" {
|
||||||
|
return defaultLoc
|
||||||
|
}
|
||||||
|
return "Moscow"
|
||||||
|
}
|
||||||
@@ -91,7 +91,10 @@ func handleEcosystem(w http.ResponseWriter, r *http.Request, urls ecoURLs) {
|
|||||||
var d ecoData
|
var d ecoData
|
||||||
var wg sync.WaitGroup
|
var wg sync.WaitGroup
|
||||||
wg.Add(3)
|
wg.Add(3)
|
||||||
go func() { defer wg.Done(); d.Nexus.Err = getEco(ctx, urls.nexus, "/api/v1/entities?limit=50", &d.Nexus.Rows) }()
|
go func() {
|
||||||
|
defer wg.Done()
|
||||||
|
d.Nexus.Err = getEco(ctx, urls.nexus, "/api/v1/entities?limit=50", &d.Nexus.Rows)
|
||||||
|
}()
|
||||||
go func() {
|
go func() {
|
||||||
defer wg.Done()
|
defer wg.Done()
|
||||||
d.Praxis.Err = getEco(ctx, urls.praxis, "/api/v1/items?limit=50", &d.Praxis.Rows)
|
d.Praxis.Err = getEco(ctx, urls.praxis, "/api/v1/items?limit=50", &d.Praxis.Rows)
|
||||||
|
|||||||
+203
-3
@@ -17,11 +17,12 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
// fakeCore records the mutating calls handleTools makes and returns canned
|
// fakeCore records the mutating calls handleTools makes and returns canned
|
||||||
// tool lists / errors. Embedding ipc.CoreAPI (nil) satisfies the large
|
// tool lists / errors. Embedding ipc.UnimplementedCoreAPI satisfies the large
|
||||||
// interface — only the methods the handlers touch are overridden; any other
|
// interface — only the methods the handlers touch are overridden; any other
|
||||||
// call would nil-panic, which is fine since the handlers never make them.
|
// call returns ipc.ErrNotImplemented instead of nil-panicking, so a test that
|
||||||
|
// accidentally exercises an undeclared method fails loudly.
|
||||||
type fakeCore struct {
|
type fakeCore struct {
|
||||||
ipc.CoreAPI
|
ipc.UnimplementedCoreAPI
|
||||||
|
|
||||||
proposed, enabled []ipc.Tool
|
proposed, enabled []ipc.Tool
|
||||||
listErr error
|
listErr error
|
||||||
@@ -62,6 +63,18 @@ type fakeCore struct {
|
|||||||
// for handleTrace tests
|
// for handleTrace tests
|
||||||
tickTrace ipc.TickTrace
|
tickTrace ipc.TickTrace
|
||||||
traceErr error
|
traceErr error
|
||||||
|
|
||||||
|
// for handleChatAPI tests
|
||||||
|
chatText string
|
||||||
|
chatErr error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeCore) Chat(_ context.Context, text string) (string, error) {
|
||||||
|
f.chatText = text
|
||||||
|
if f.chatErr != nil {
|
||||||
|
return "", f.chatErr
|
||||||
|
}
|
||||||
|
return "поняла", nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func (f *fakeCore) EnableTool(_ context.Context, name string, cmd []string, destructive bool, scope string, _ time.Time) error {
|
func (f *fakeCore) EnableTool(_ context.Context, name string, cmd []string, destructive bool, scope string, _ time.Time) error {
|
||||||
@@ -918,3 +931,190 @@ func TestHandleTools_ListToolsError_502(t *testing.T) {
|
|||||||
t.Fatalf("status = %d, want 502; body=%s", rr.Code, rr.Body.String())
|
t.Fatalf("status = %d, want 502; body=%s", rr.Code, rr.Body.String())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- handleRoutines ---
|
||||||
|
|
||||||
|
// routineCore is a fakeCore that also answers the proposed-routine calls.
|
||||||
|
type routineCore struct {
|
||||||
|
fakeCore
|
||||||
|
|
||||||
|
routines []ipc.ProposedRoutine
|
||||||
|
dismissed int64
|
||||||
|
acceptedID int64
|
||||||
|
remCron string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *routineCore) ListProposedRoutines(_ context.Context) ([]ipc.ProposedRoutine, error) {
|
||||||
|
return c.routines, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *routineCore) DismissProposedRoutine(_ context.Context, id int64) error {
|
||||||
|
c.dismissed = id
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *routineCore) AcceptProposedRoutine(_ context.Context, id int64) error {
|
||||||
|
c.acceptedID = id
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *routineCore) CreateReminder(_ context.Context, _ time.Time, _, cron string) (int64, error) {
|
||||||
|
c.remCron = cron
|
||||||
|
return 77, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func weeklyRoutineCore() *routineCore {
|
||||||
|
return &routineCore{routines: []ipc.ProposedRoutine{{
|
||||||
|
ID: 3, Action: "refill", Object: "cat_water", IntervalDays: 7,
|
||||||
|
Status: "proposed", CreatedTs: time.Now().Add(-2 * time.Hour).UnixMilli(),
|
||||||
|
}}}
|
||||||
|
}
|
||||||
|
|
||||||
|
func postRoutine(action, id string) *http.Request {
|
||||||
|
return postForm(action, url.Values{"id": {id}})
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleRoutines_GET_ShowsMavensPhrase(t *testing.T) {
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, httptest.NewRequest(http.MethodGet, "/routines", nil), weeklyRoutineCore(), nil, false)
|
||||||
|
if rr.Code != http.StatusOK {
|
||||||
|
t.Fatalf("status = %d, want 200", rr.Code)
|
||||||
|
}
|
||||||
|
body := rr.Body.String()
|
||||||
|
if !strings.Contains(body, "заправляешь") {
|
||||||
|
t.Fatalf("want maven's phrasing in the page, got: %s", body)
|
||||||
|
}
|
||||||
|
if !strings.Contains(body, "class=scroll") {
|
||||||
|
t.Fatal("table must be wrapped in <div class=scroll> so it pans on a phone")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Accepting hands the loop a new reason to speak, so it needs step-up.
|
||||||
|
func TestHandleRoutines_Accept_RequiresStepUp(t *testing.T) {
|
||||||
|
core := weeklyRoutineCore()
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, postRoutine("accept", "3"), core, webauthn.NewPasskeySession(5*time.Minute), false)
|
||||||
|
if rr.Code != http.StatusForbidden {
|
||||||
|
t.Fatalf("status = %d, want 403", rr.Code)
|
||||||
|
}
|
||||||
|
if core.acceptedID != 0 {
|
||||||
|
t.Fatal("accepted without step-up")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Accepting only flips the status. It used to also create a one-shot reminder,
|
||||||
|
// which is why a non-weekly routine fired once and then went quiet forever
|
||||||
|
// (Vikunja #366). The tick loop owns the schedule now, so a reminder here would
|
||||||
|
// be a second, competing schedule.
|
||||||
|
func TestHandleRoutines_Accept_FlipsStatusAndMakesNoReminder(t *testing.T) {
|
||||||
|
core := weeklyRoutineCore()
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, postRoutine("accept", "3"), core, stepUpSession(), false)
|
||||||
|
if rr.Code != http.StatusOK {
|
||||||
|
t.Fatalf("status = %d, want 200; body=%s", rr.Code, rr.Body.String())
|
||||||
|
}
|
||||||
|
if core.acceptedID != 3 {
|
||||||
|
t.Fatalf("accepted id = %d, want 3", core.acceptedID)
|
||||||
|
}
|
||||||
|
if core.remCron != "" {
|
||||||
|
t.Fatalf("accepting must not create a reminder, got cron %q", core.remCron)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Dismiss only ever removes a reason to speak, so it is not step-up gated.
|
||||||
|
func TestHandleRoutines_Dismiss_NoStepUpNeeded(t *testing.T) {
|
||||||
|
core := weeklyRoutineCore()
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, postRoutine("dismiss", "3"), core, webauthn.NewPasskeySession(5*time.Minute), false)
|
||||||
|
if rr.Code != http.StatusOK {
|
||||||
|
t.Fatalf("status = %d, want 200; body=%s", rr.Code, rr.Body.String())
|
||||||
|
}
|
||||||
|
if core.dismissed != 3 {
|
||||||
|
t.Fatalf("dismissed = %d, want 3", core.dismissed)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleRoutines_UnknownAction_400(t *testing.T) {
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, postRoutine("frobnicate", "3"), weeklyRoutineCore(), stepUpSession(), false)
|
||||||
|
if rr.Code != http.StatusBadRequest {
|
||||||
|
t.Fatalf("status = %d, want 400", rr.Code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleRoutines_BadID_400(t *testing.T) {
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, postRoutine("dismiss", "nope"), weeklyRoutineCore(), stepUpSession(), false)
|
||||||
|
if rr.Code != http.StatusBadRequest {
|
||||||
|
t.Fatalf("status = %d, want 400", rr.Code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleRoutines_NilCore_503(t *testing.T) {
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleRoutines(rr, httptest.NewRequest(http.MethodGet, "/routines", nil), nil, nil, false)
|
||||||
|
if rr.Code != http.StatusServiceUnavailable {
|
||||||
|
t.Fatalf("status = %d, want 503", rr.Code)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --- handleChatAPI step-up gate (Vikunja #317) ---
|
||||||
|
//
|
||||||
|
// POST /api/chat reaches the router, the LLM and the act path, so it carries
|
||||||
|
// the same gate as POST /tools and POST /api/revert.
|
||||||
|
|
||||||
|
func postChat(text string) *http.Request {
|
||||||
|
req := httptest.NewRequest(http.MethodPost, "/api/chat", strings.NewReader("text="+url.QueryEscape(text)))
|
||||||
|
req.Header.Set("Content-Type", "application/x-www-form-urlencoded")
|
||||||
|
return req
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleChatAPI_RequireStepUp_FailsClosed(t *testing.T) {
|
||||||
|
core := &fakeCore{}
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleChatAPI(rr, postChat("выключи свет"), core, nil, true)
|
||||||
|
if rr.Code != http.StatusForbidden {
|
||||||
|
t.Fatalf("status = %d, want 403; body=%s", rr.Code, rr.Body.String())
|
||||||
|
}
|
||||||
|
if core.chatText != "" {
|
||||||
|
t.Errorf("core.Chat called with %q, but -require-stepup should deny", core.chatText)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleChatAPI_UnassertedSession_Denied(t *testing.T) {
|
||||||
|
core := &fakeCore{}
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleChatAPI(rr, postChat("выключи свет"), core, webauthn.NewPasskeySession(5*time.Minute), false)
|
||||||
|
if rr.Code != http.StatusForbidden {
|
||||||
|
t.Fatalf("status = %d, want 403", rr.Code)
|
||||||
|
}
|
||||||
|
if core.chatText != "" {
|
||||||
|
t.Errorf("core.Chat called with %q despite an unasserted session", core.chatText)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestHandleChatAPI_AssertedSession_PassesGate(t *testing.T) {
|
||||||
|
core := &fakeCore{}
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleChatAPI(rr, postChat("привет"), core, stepUpSession(), true)
|
||||||
|
if rr.Code != http.StatusSeeOther {
|
||||||
|
t.Fatalf("status = %d, want 303; body=%s", rr.Code, rr.Body.String())
|
||||||
|
}
|
||||||
|
if core.chatText != "привет" {
|
||||||
|
t.Errorf("core.Chat text = %q, want %q", core.chatText, "привет")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Default deploy: WebAuthn unconfigured and -require-stepup off ⇒ chat keeps
|
||||||
|
// working, resting on the transport-level auth in front of mavweb.
|
||||||
|
func TestHandleChatAPI_FailOpenByDefault(t *testing.T) {
|
||||||
|
core := &fakeCore{}
|
||||||
|
rr := httptest.NewRecorder()
|
||||||
|
handleChatAPI(rr, postChat("привет"), core, nil, false)
|
||||||
|
if rr.Code != http.StatusSeeOther {
|
||||||
|
t.Fatalf("status = %d, want 303", rr.Code)
|
||||||
|
}
|
||||||
|
if core.chatText != "привет" {
|
||||||
|
t.Errorf("core.Chat text = %q, want %q", core.chatText, "привет")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+124
-24
@@ -24,6 +24,7 @@ import (
|
|||||||
"github.com/coder/websocket"
|
"github.com/coder/websocket"
|
||||||
"github.com/kami/maven/internal/audio"
|
"github.com/kami/maven/internal/audio"
|
||||||
"github.com/kami/maven/internal/ipc"
|
"github.com/kami/maven/internal/ipc"
|
||||||
|
"github.com/kami/maven/internal/pattern"
|
||||||
"github.com/kami/maven/internal/voice"
|
"github.com/kami/maven/internal/voice"
|
||||||
"github.com/kami/maven/internal/webauthn"
|
"github.com/kami/maven/internal/webauthn"
|
||||||
)
|
)
|
||||||
@@ -313,7 +314,7 @@ func noCache(h http.Handler) http.Handler {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
addr := flag.String("addr", ":9200", "HTTP listen address")
|
addr := flag.String("addr", "127.0.0.1:9200", "HTTP listen address (loopback by default; pass e.g. \":9200\" or a LAN IP deliberately for wider exposure — POST /chat and /routines are state-changing)")
|
||||||
voiceAddr := flag.String("voice", "127.0.0.1:9100", "voice server TCP addr (host:port)")
|
voiceAddr := flag.String("voice", "127.0.0.1:9100", "voice server TCP addr (host:port)")
|
||||||
// ntfyWS: the ntfy WebSocket subscribe URL the PWA connects to for in-app
|
// ntfyWS: the ntfy WebSocket subscribe URL the PWA connects to for in-app
|
||||||
// nudge delivery, e.g. wss://ntfy.kvmx.ru/maven/ws?auth=<base64-token>. The
|
// nudge delivery, e.g. wss://ntfy.kvmx.ru/maven/ws?auth=<base64-token>. The
|
||||||
@@ -328,7 +329,7 @@ func main() {
|
|||||||
coreSock := flag.String("core", "", "mavend IPC socket path for presence-signal ingest (empty = disabled)")
|
coreSock := flag.String("core", "", "mavend IPC socket path for presence-signal ingest (empty = disabled)")
|
||||||
pkOrigin := flag.String("webauthn-origin", "", "WebAuthn origin URL (e.g. https://maven.kvmx.ru)")
|
pkOrigin := flag.String("webauthn-origin", "", "WebAuthn origin URL (e.g. https://maven.kvmx.ru)")
|
||||||
pkRPID := flag.String("webauthn-rpid", "", "WebAuthn RP ID (e.g. maven.kvmx.ru)")
|
pkRPID := flag.String("webauthn-rpid", "", "WebAuthn RP ID (e.g. maven.kvmx.ru)")
|
||||||
requireStepUp := flag.Bool("require-stepup", false, "fail closed on step-up-gated actions (/tools POST, /api/revert) when WebAuthn step-up cannot be asserted; default false preserves the historical fail-open behaviour")
|
requireStepUp := flag.Bool("require-stepup", false, "fail closed on step-up-gated actions (POST /tools, /routines, /api/revert, /api/chat) when WebAuthn step-up cannot be asserted; default false preserves the historical fail-open behaviour")
|
||||||
pkFile := flag.String("passkey-file", "./passkeys.json", "path to WebAuthn credential store (JSON)")
|
pkFile := flag.String("passkey-file", "./passkeys.json", "path to WebAuthn credential store (JSON)")
|
||||||
nexusURL := flag.String("nexus", "", "Nexus base URL for the /ecosystem panel (empty = not configured)")
|
nexusURL := flag.String("nexus", "", "Nexus base URL for the /ecosystem panel (empty = not configured)")
|
||||||
praxisURL := flag.String("praxis", "", "Praxis base URL for the /ecosystem panel (empty = not configured)")
|
praxisURL := flag.String("praxis", "", "Praxis base URL for the /ecosystem panel (empty = not configured)")
|
||||||
@@ -395,9 +396,6 @@ func main() {
|
|||||||
mux.HandleFunc("/reminders", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/reminders", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleReminders(w, r, core)
|
handleReminders(w, r, core)
|
||||||
})
|
})
|
||||||
mux.HandleFunc("/routines", func(w http.ResponseWriter, r *http.Request) {
|
|
||||||
handleRoutines(w, r, core)
|
|
||||||
})
|
|
||||||
mux.HandleFunc("/morning", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/morning", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleMorning(w, r, core)
|
handleMorning(w, r, core)
|
||||||
})
|
})
|
||||||
@@ -436,9 +434,9 @@ func main() {
|
|||||||
}
|
}
|
||||||
if stepUpSession == nil {
|
if stepUpSession == nil {
|
||||||
if *requireStepUp {
|
if *requireStepUp {
|
||||||
log.Printf("SECURITY: step-up verification is DISABLED (-webauthn-origin/-webauthn-rpid unset) and -require-stepup is set: POST /tools (tool enable/disable/dismiss — defines and executes arbitrary argv) and POST /api/revert will be DENIED (403). Set -webauthn-origin and -webauthn-rpid to enable passkey step-up.")
|
log.Printf("SECURITY: step-up verification is DISABLED (-webauthn-origin/-webauthn-rpid unset) and -require-stepup is set: POST /tools (tool enable/disable/dismiss — defines and executes arbitrary argv), POST /routines (accepting schedules recurring firing), POST /api/revert and POST /api/chat (reaches the router, the LLM and the act path) will be DENIED (403). Set -webauthn-origin and -webauthn-rpid to enable passkey step-up.")
|
||||||
} else {
|
} else {
|
||||||
log.Printf("SECURITY WARNING: step-up verification is DISABLED because -webauthn-origin/-webauthn-rpid are unset. UNGUARDED SURFACES: POST /tools (defines arbitrary argv via name+cmd, which internal/tool then EXECUTES) and POST /api/revert (voids the latest fact for a key). These are protected only by whatever transport-level auth sits in front of mavweb (wg+nginx+auth) — do NOT expose -addr on a public interface. Set -webauthn-origin and -webauthn-rpid to require passkey step-up, or pass -require-stepup to fail closed instead.")
|
log.Printf("SECURITY WARNING: step-up verification is DISABLED because -webauthn-origin/-webauthn-rpid are unset. UNGUARDED SURFACES: POST /tools (defines arbitrary argv via name+cmd, which internal/tool then EXECUTES), POST /routines (accepting schedules recurring firing), POST /api/revert (voids the latest fact for a key) and POST /api/chat (reaches the router, the LLM and, through applyAction, the act path). These are protected only by whatever transport-level auth sits in front of mavweb (wg+nginx+auth) — do NOT expose -addr on a public interface. Set -webauthn-origin and -webauthn-rpid to require passkey step-up, or pass -require-stepup to fail closed instead.")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -450,15 +448,32 @@ func main() {
|
|||||||
mux.HandleFunc("/tools", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/tools", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleTools(w, r, core, stepUpSession, *requireStepUp)
|
handleTools(w, r, core, stepUpSession, *requireStepUp)
|
||||||
})
|
})
|
||||||
|
// /routines — the authed accept surface. Registered here, next to /tools,
|
||||||
|
// because accepting shares the same step-up gate.
|
||||||
|
mux.HandleFunc("/routines", func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
handleRoutines(w, r, core, stepUpSession, *requireStepUp)
|
||||||
|
})
|
||||||
|
|
||||||
// /api/revert voids the latest fact for a key — a store mutation, so it
|
// State-changing routes on this server, and their gate (Vikunja #317):
|
||||||
// sits behind the same passkey step-up as tool enable (nil session ⇒
|
//
|
||||||
// WebAuthn unconfigured ⇒ transport-level auth only, same as /tools).
|
// POST /tools step-up — defines argv that internal/tool executes
|
||||||
|
// POST /routines step-up — accepting schedules recurring firing
|
||||||
|
// POST /api/revert step-up — voids the latest fact for a key
|
||||||
|
// POST /api/chat step-up — reaches the router, LLM and the act path
|
||||||
|
// POST /api/signal none — appends a presence fact, no argv, no act
|
||||||
|
// POST /api/ptt, /ws none — proxy audio to mavend's voice port, which
|
||||||
|
// is itself only reachable inside the deploy
|
||||||
|
//
|
||||||
|
// "step-up" means stepUpOK: asserted passkey when WebAuthn is configured,
|
||||||
|
// otherwise fail-open unless -require-stepup, which denies.
|
||||||
|
//
|
||||||
|
// GET /chat only renders the page and echoes back the q/r query params the
|
||||||
|
// POST redirect set — nothing to gate.
|
||||||
mux.HandleFunc("/chat", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/chat", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleChatPage(w, r, core)
|
handleChatPage(w, r, core)
|
||||||
})
|
})
|
||||||
mux.HandleFunc("/api/chat", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/api/chat", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleChatAPI(w, r, core)
|
handleChatAPI(w, r, core, stepUpSession, *requireStepUp)
|
||||||
})
|
})
|
||||||
mux.HandleFunc("/api/revert", func(w http.ResponseWriter, r *http.Request) {
|
mux.HandleFunc("/api/revert", func(w http.ResponseWriter, r *http.Request) {
|
||||||
handleRevert(w, r, core, stepUpSession, *requireStepUp)
|
handleRevert(w, r, core, stepUpSession, *requireStepUp)
|
||||||
@@ -680,22 +695,24 @@ const toolsHTML = `{{template "shellTop" "tools"}}
|
|||||||
</section>
|
</section>
|
||||||
{{template "shellBottom"}}`
|
{{template "shellBottom"}}`
|
||||||
|
|
||||||
// routinesHTML — proposed routine review surface. Lists detected patterns
|
// routinesHTML — proposed routine review surface. One row per thing maven
|
||||||
// awaiting human confirmation, with accept (→ reminder) and dismiss buttons.
|
// noticed, in her words, with at most two actions: accept or dismiss.
|
||||||
const routinesHTML = `{{template "shellTop" "routines"}}
|
const routinesHTML = `{{template "shellTop" "routines"}}
|
||||||
<h1>Routines</h1>
|
<h1>Routines</h1>
|
||||||
{{if .Msg}}<div class="msg msg-ok">{{.Msg}}</div>{{end}}
|
{{if .Msg}}<div class="msg msg-ok">{{.Msg}}</div>{{end}}
|
||||||
<section class=card>
|
<section class=card>
|
||||||
<h2 class=card-title>proposed <span class=badge>{{len .Proposed}}</span></h2>
|
<h2 class=card-title>noticed <span class=badge>{{len .Proposed}}</span></h2>
|
||||||
{{if .Proposed}}<div class=scroll><table><tr><th>action</th><th>object</th><th>every</th><th></th></tr>
|
{{if .Proposed}}<div class=scroll><table><tr><th>maven noticed</th><th>when</th><th></th><th></th></tr>
|
||||||
{{range .Proposed}}<tr>
|
{{range .Proposed}}<tr>
|
||||||
<td><code>{{.Action}}</code></td><td><code>{{.Object}}</code></td><td>{{.IntervalDays}} days</td>
|
<td>{{.Phrase}}</td><td class=muted>{{.Noticed}}</td>
|
||||||
<td>
|
<td><form method=post action=/routines class=inline-form>
|
||||||
<form method=post action=/routines class=inline-form>
|
<input type=hidden name=id value="{{.ID}}">
|
||||||
|
<input type=hidden name=action value=accept>
|
||||||
|
<button class=btn>accept</button></form></td>
|
||||||
|
<td><form method=post action=/routines class=inline-form>
|
||||||
<input type=hidden name=id value="{{.ID}}">
|
<input type=hidden name=id value="{{.ID}}">
|
||||||
<input type=hidden name=action value=dismiss>
|
<input type=hidden name=action value=dismiss>
|
||||||
<button class="btn btn-muted">dismiss</button></form>
|
<button class="btn btn-muted">dismiss</button></form></td>
|
||||||
</td>
|
|
||||||
</tr>{{end}}</table></div>
|
</tr>{{end}}</table></div>
|
||||||
{{else}}<div class=empty>
|
{{else}}<div class=empty>
|
||||||
<svg class=icon width="20" height="20"><use href="/ethos-icons.svg#i-wave"/></svg>
|
<svg class=icon width="20" height="20"><use href="/ethos-icons.svg#i-wave"/></svg>
|
||||||
@@ -785,7 +802,23 @@ func handleReminders(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func handleRoutines(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
// routineRow is one line on the page: what maven noticed, in her words, and
|
||||||
|
// how long ago she noticed it.
|
||||||
|
type routineRow struct {
|
||||||
|
ID int64
|
||||||
|
Phrase string
|
||||||
|
Noticed string
|
||||||
|
}
|
||||||
|
|
||||||
|
// handleRoutines serves the routine review surface (GET) and answers a
|
||||||
|
// proposal (POST id + action=accept|dismiss).
|
||||||
|
//
|
||||||
|
// Accept is gated at step-up, the same tier as enabling a tool: saying yes
|
||||||
|
// hands the trigger loop a new standing reason to speak to the human, so it
|
||||||
|
// moves the boundary and only an authed surface may do it. Dismiss is not
|
||||||
|
// gated — it only ever removes a reason to speak, so the worst a weaker caller
|
||||||
|
// can do is make maven quieter.
|
||||||
|
func handleRoutines(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI, session *webauthn.PasskeySession, requireStepUp bool) {
|
||||||
if core == nil {
|
if core == nil {
|
||||||
http.Error(w, "routines disabled (no -core)", http.StatusServiceUnavailable)
|
http.Error(w, "routines disabled (no -core)", http.StatusServiceUnavailable)
|
||||||
return
|
return
|
||||||
@@ -801,6 +834,17 @@ func handleRoutines(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
|||||||
return
|
return
|
||||||
}
|
}
|
||||||
switch action {
|
switch action {
|
||||||
|
case "accept":
|
||||||
|
if !stepUpOK(session, requireStepUp) {
|
||||||
|
http.Error(w, "step-up required: assert a passkey first", http.StatusForbidden)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err := acceptRoutine(ctx, core, rid); err != nil {
|
||||||
|
log.Printf("routines: accept %d: %v", rid, err)
|
||||||
|
http.Error(w, "accept failed: "+err.Error(), http.StatusBadGateway)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
msg = "accepted routine — maven will remind you"
|
||||||
case "dismiss":
|
case "dismiss":
|
||||||
if err := core.DismissProposedRoutine(ctx, rid); err != nil {
|
if err := core.DismissProposedRoutine(ctx, rid); err != nil {
|
||||||
log.Printf("routines: dismiss %d: %v", rid, err)
|
log.Printf("routines: dismiss %d: %v", rid, err)
|
||||||
@@ -822,12 +866,57 @@ func handleRoutines(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
|||||||
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
w.Header().Set("Content-Type", "text/html; charset=utf-8")
|
||||||
if err := routinesTmpl.Execute(w, struct {
|
if err := routinesTmpl.Execute(w, struct {
|
||||||
Msg string
|
Msg string
|
||||||
Proposed []ipc.ProposedRoutine
|
Proposed []routineRow
|
||||||
}{msg, proposed}); err != nil {
|
}{msg, routineRows(proposed)}); err != nil {
|
||||||
log.Printf("routines render: %v", err)
|
log.Printf("routines render: %v", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// routineRows turns the wire rows into display rows. The phrase comes from
|
||||||
|
// pattern.PhraseRoutine so the page says the same thing maven's voice says.
|
||||||
|
func routineRows(rs []ipc.ProposedRoutine) []routineRow {
|
||||||
|
out := make([]routineRow, 0, len(rs))
|
||||||
|
for _, r := range rs {
|
||||||
|
p := pattern.ProposedRoutine{Action: r.Action, Object: r.Object, IntervalDays: r.IntervalDays}
|
||||||
|
noticed := "just now"
|
||||||
|
if r.CreatedTs > 0 {
|
||||||
|
noticed = time.Since(time.UnixMilli(r.CreatedTs)).Round(time.Minute).String() + " ago"
|
||||||
|
}
|
||||||
|
out = append(out, routineRow{ID: r.ID, Phrase: pattern.PhraseRoutine(&p), Noticed: noticed})
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// acceptRoutine creates the recurring reminder for a proposal, then marks the
|
||||||
|
// proposal accepted and links the reminder to it. Weekly patterns get a cron
|
||||||
|
// expression; any other interval fires once.
|
||||||
|
//
|
||||||
|
// TODO(vikunja#46): this mirrors the voice accept path in cmd/mavend/voice.go.
|
||||||
|
// When the tick loop learns to read accepted proposals directly, both callers
|
||||||
|
// should hand off to one place in core instead of each building a reminder.
|
||||||
|
func acceptRoutine(ctx context.Context, core ipc.CoreAPI, id int64) error {
|
||||||
|
proposed, err := core.ListProposedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
var found *ipc.ProposedRoutine
|
||||||
|
for i := range proposed {
|
||||||
|
if proposed[i].ID == id {
|
||||||
|
found = &proposed[i]
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if found == nil {
|
||||||
|
return errors.New("no such proposed routine")
|
||||||
|
}
|
||||||
|
|
||||||
|
// No reminder is created here. Accepting only flips the status; the tick
|
||||||
|
// loop reads accepted routines and nudges on the interval (Vikunja #366).
|
||||||
|
// The old code made a one-shot reminder, so a non-weekly routine fired
|
||||||
|
// once and then went quiet forever.
|
||||||
|
return core.AcceptProposedRoutine(ctx, id)
|
||||||
|
}
|
||||||
|
|
||||||
func handleTrace(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
func handleTrace(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
||||||
if core == nil {
|
if core == nil {
|
||||||
http.Error(w, "trace disabled (no -core)", http.StatusServiceUnavailable)
|
http.Error(w, "trace disabled (no -core)", http.StatusServiceUnavailable)
|
||||||
@@ -1181,7 +1270,14 @@ func handleChatPage(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// handleChatAPI processes a chat message POST and redirects back to /chat.
|
// handleChatAPI processes a chat message POST and redirects back to /chat.
|
||||||
func handleChatAPI(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
//
|
||||||
|
// State-changing, and the widest surface on this server: the text reaches the
|
||||||
|
// router, the LLM, and through mavend's applyAction the whole action path
|
||||||
|
// including `act` — so it is gated on the same step-up as POST /tools and
|
||||||
|
// POST /api/revert (Vikunja #317). With WebAuthn unconfigured the gate is
|
||||||
|
// fail-open exactly like the others (see stepUpOK); with -require-stepup it
|
||||||
|
// denies, which is the point of that flag.
|
||||||
|
func handleChatAPI(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI, session *webauthn.PasskeySession, requireStepUp bool) {
|
||||||
if r.Method != http.MethodPost {
|
if r.Method != http.MethodPost {
|
||||||
http.Error(w, "POST only", http.StatusMethodNotAllowed)
|
http.Error(w, "POST only", http.StatusMethodNotAllowed)
|
||||||
return
|
return
|
||||||
@@ -1190,6 +1286,10 @@ func handleChatAPI(w http.ResponseWriter, r *http.Request, core ipc.CoreAPI) {
|
|||||||
http.Error(w, "chat disabled (no -core)", http.StatusServiceUnavailable)
|
http.Error(w, "chat disabled (no -core)", http.StatusServiceUnavailable)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if !stepUpOK(session, requireStepUp) {
|
||||||
|
http.Error(w, "step-up required: assert a passkey first", http.StatusForbidden)
|
||||||
|
return
|
||||||
|
}
|
||||||
text := strings.TrimSpace(r.FormValue("text"))
|
text := strings.TrimSpace(r.FormValue("text"))
|
||||||
if text == "" {
|
if text == "" {
|
||||||
http.Redirect(w, r, "/chat", http.StatusSeeOther)
|
http.Redirect(w, r, "/chat", http.StatusSeeOther)
|
||||||
|
|||||||
@@ -8,6 +8,18 @@
|
|||||||
#
|
#
|
||||||
# Maven's own compose joins this same network (add `ecosystem` as an external
|
# Maven's own compose joins this same network (add `ecosystem` as an external
|
||||||
# network there) to reach nexus:9740 / praxis:8989 / hexis:9741 directly.
|
# network there) to reach nexus:9740 / praxis:8989 / hexis:9741 directly.
|
||||||
|
#
|
||||||
|
# NO RELEASE PINNING (Vikunja #354): each `build:` below points at a sibling
|
||||||
|
# WORKING TREE, so `up --build` ships whatever is checked out there, including
|
||||||
|
# uncommitted edits. Before bringing this up, check what you are about to
|
||||||
|
# deploy:
|
||||||
|
#
|
||||||
|
# for r in nexus praxis hexis; do git -C ../../../$r status --short; \
|
||||||
|
# git -C ../../../$r log -1 --oneline; done
|
||||||
|
#
|
||||||
|
# The host nginx that fronts these is deploy/ecosystem/nginx.conf — it binds
|
||||||
|
# the wg and LAN addresses only, with allow/deny. Keep it that way: none of
|
||||||
|
# these containers has auth of its own.
|
||||||
name: ecosystem
|
name: ecosystem
|
||||||
|
|
||||||
services:
|
services:
|
||||||
|
|||||||
@@ -1,13 +1,71 @@
|
|||||||
# Reverse-proxy the three sibling admin UIs. Drop into your nginx sites (or the
|
# Reverse-proxy Maven's own web UI plus the three sibling admin UIs. Drop into
|
||||||
# nginx-panel app) and reload. Assumes the compose publishes each service on
|
# your nginx sites (or the nginx-panel app) and reload. Assumes the compose
|
||||||
# 127.0.0.1:<port>. Add TLS (certbot / your existing cert block) per server.
|
# publishes each service on 127.0.0.1:<port>. Add TLS (certbot / your existing
|
||||||
|
# cert block) per server.
|
||||||
#
|
#
|
||||||
# NOTE: hexis.<domain> previously pointed at the MCP tool — repoint that
|
# NOTE: hexis.<domain> previously pointed at the MCP tool — repoint that
|
||||||
# elsewhere first (the app now owns hexis.*).
|
# elsewhere first (the app now owns hexis.*).
|
||||||
|
#
|
||||||
|
# 10.42.0.1 and 192.168.1.104 below are THIS BOX's WireGuard and LAN
|
||||||
|
# addresses (homesrv) — these admin UIs have no auth of their own, so the
|
||||||
|
# explicit bind + allow/deny below is what keeps them off the open internet.
|
||||||
|
# On a different box, replace both addresses with that box's wg and LAN IPs.
|
||||||
|
# Do NOT "fix" a failed bind by reverting to `listen 80` (all interfaces) —
|
||||||
|
# that removes the only access control these containers have.
|
||||||
|
|
||||||
|
# maven.<domain> → mavweb (docker-compose.yml publishes it on 127.0.0.1:9201).
|
||||||
|
# Same bind + ACL as the siblings, and for a stronger reason: mavweb serves
|
||||||
|
# POST /tools, which defines argv that internal/tool EXECUTES, plus POST
|
||||||
|
# /routines, /api/revert and /api/chat (Vikunja #317). Without
|
||||||
|
# -webauthn-origin/-webauthn-rpid mavweb has no auth of its own, so this block
|
||||||
|
# is the auth. If you add TLS and a basic-auth/oauth2-proxy layer, keep the
|
||||||
|
# allow/deny anyway — belt and braces on an RCE surface.
|
||||||
|
#
|
||||||
|
# WebSocket upgrade matters here: /ws carries push-to-talk audio, so the
|
||||||
|
# Upgrade/Connection headers below are required, not decoration. The map keeps
|
||||||
|
# `Connection: upgrade` off plain requests; it sits in the http context, which
|
||||||
|
# is where sites-available files are included — if your nginx already defines
|
||||||
|
# $connection_upgrade, drop this block.
|
||||||
|
map $http_upgrade $connection_upgrade {
|
||||||
|
default upgrade;
|
||||||
|
'' close;
|
||||||
|
}
|
||||||
|
|
||||||
server {
|
server {
|
||||||
listen 80;
|
listen 10.42.0.1:80;
|
||||||
|
listen 192.168.1.104:80;
|
||||||
|
server_name maven.kvmx.ru;
|
||||||
|
|
||||||
|
allow 10.42.0.0/24;
|
||||||
|
allow 192.168.1.0/24;
|
||||||
|
deny all;
|
||||||
|
|
||||||
|
# push-to-talk uploads raw PCM; the default 1m is enough for a short
|
||||||
|
# utterance but not for a long one.
|
||||||
|
client_max_body_size 32m;
|
||||||
|
|
||||||
|
location / {
|
||||||
|
proxy_pass http://127.0.0.1:9201;
|
||||||
|
proxy_http_version 1.1;
|
||||||
|
proxy_set_header Upgrade $http_upgrade;
|
||||||
|
proxy_set_header Connection $connection_upgrade;
|
||||||
|
proxy_set_header Host $host;
|
||||||
|
proxy_set_header X-Real-IP $remote_addr;
|
||||||
|
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||||
|
proxy_set_header X-Forwarded-Proto $scheme;
|
||||||
|
proxy_read_timeout 300s; # an LLM turn can take minutes on the iGPU
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
server {
|
||||||
|
listen 10.42.0.1:80;
|
||||||
|
listen 192.168.1.104:80;
|
||||||
server_name nexus.kvmx.ru;
|
server_name nexus.kvmx.ru;
|
||||||
|
|
||||||
|
allow 10.42.0.0/24;
|
||||||
|
allow 192.168.1.0/24;
|
||||||
|
deny all;
|
||||||
|
|
||||||
location / {
|
location / {
|
||||||
proxy_pass http://127.0.0.1:9740;
|
proxy_pass http://127.0.0.1:9740;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
@@ -18,8 +76,14 @@ server {
|
|||||||
}
|
}
|
||||||
|
|
||||||
server {
|
server {
|
||||||
listen 80;
|
listen 10.42.0.1:80;
|
||||||
|
listen 192.168.1.104:80;
|
||||||
server_name praxis.kvmx.ru;
|
server_name praxis.kvmx.ru;
|
||||||
|
|
||||||
|
allow 10.42.0.0/24;
|
||||||
|
allow 192.168.1.0/24;
|
||||||
|
deny all;
|
||||||
|
|
||||||
location / {
|
location / {
|
||||||
proxy_pass http://127.0.0.1:8989;
|
proxy_pass http://127.0.0.1:8989;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
@@ -30,8 +94,14 @@ server {
|
|||||||
}
|
}
|
||||||
|
|
||||||
server {
|
server {
|
||||||
listen 80;
|
listen 10.42.0.1:80;
|
||||||
|
listen 192.168.1.104:80;
|
||||||
server_name hexis.kvmx.ru;
|
server_name hexis.kvmx.ru;
|
||||||
|
|
||||||
|
allow 10.42.0.0/24;
|
||||||
|
allow 192.168.1.0/24;
|
||||||
|
deny all;
|
||||||
|
|
||||||
location / {
|
location / {
|
||||||
proxy_pass http://127.0.0.1:9741;
|
proxy_pass http://127.0.0.1:9741;
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
|
|||||||
+15
-5
@@ -6,11 +6,12 @@
|
|||||||
"state_dir": "/var/lib/maven",
|
"state_dir": "/var/lib/maven",
|
||||||
|
|
||||||
"phraser": {
|
"phraser": {
|
||||||
"model_path": "/opt/maven/models/llm/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf",
|
"model_path": "/opt/maven/models/llm/qwen3/Qwen3-1.7B-UD-Q4_K_XL.gguf",
|
||||||
"bin_path": "llama-server",
|
"bin_path": "llama-server",
|
||||||
"n_gpu_layers": 99,
|
"n_gpu_layers": 99,
|
||||||
"n_ctx": 2048,
|
"n_ctx": 4096,
|
||||||
"timeout": "60s"
|
"timeout": "60s",
|
||||||
|
"llm_nudges": false
|
||||||
},
|
},
|
||||||
|
|
||||||
"telegram": {
|
"telegram": {
|
||||||
@@ -25,6 +26,11 @@
|
|||||||
"severity_ceiling": 2
|
"severity_ceiling": 2
|
||||||
},
|
},
|
||||||
|
|
||||||
|
"pattern_proposals": {
|
||||||
|
"notify": false,
|
||||||
|
"cooldown": "24h"
|
||||||
|
},
|
||||||
|
|
||||||
"nexus": { "url": "http://nexus:9740" },
|
"nexus": { "url": "http://nexus:9740" },
|
||||||
"praxis": { "url": "http://praxis:8989" },
|
"praxis": { "url": "http://praxis:8989" },
|
||||||
"hexis": { "url": "http://hexis:9741" },
|
"hexis": { "url": "http://hexis:9741" },
|
||||||
@@ -36,10 +42,14 @@
|
|||||||
"stt": { "socket": "/run/maven/stt.sock", "lang": "ru" },
|
"stt": { "socket": "/run/maven/stt.sock", "lang": "ru" },
|
||||||
"tts": { "socket": "/run/maven/tts.sock", "lang": "ru" },
|
"tts": { "socket": "/run/maven/tts.sock", "lang": "ru" },
|
||||||
"embedder": {
|
"embedder": {
|
||||||
"model_path": "/opt/maven/models/embedder/model.onnx",
|
"model_path": "/opt/maven/models/embedder/multilingual-e5-small/model_quantized.onnx",
|
||||||
"tokenizer_path": "/opt/maven/models/embedder/tokenizer.json",
|
"tokenizer_path": "/opt/maven/models/embedder/multilingual-e5-small/tokenizer.json",
|
||||||
"lib_path": "/opt/maven/lib/libonnxruntime.so"
|
"lib_path": "/opt/maven/lib/libonnxruntime.so"
|
||||||
},
|
},
|
||||||
|
"llm_router": true,
|
||||||
|
"query_min_score": 0.55,
|
||||||
|
"query_min_margin": 0.008,
|
||||||
|
"clarify_max_attempts": 3,
|
||||||
"tool_timeout": "30s",
|
"tool_timeout": "30s",
|
||||||
"tools": [
|
"tools": [
|
||||||
{ "name": "status", "cmd": ["systemctl", "status"], "scope": "homelab", "destructive": false },
|
{ "name": "status", "cmd": ["systemctl", "status"], "scope": "homelab", "destructive": false },
|
||||||
|
|||||||
@@ -25,3 +25,46 @@
|
|||||||
6. Add `/eval` API method to `ipc.CoreAPI` (or reuse `Chat` with system context) so mavweb can show evaluation history
|
6. Add `/eval` API method to `ipc.CoreAPI` (or reuse `Chat` with system context) so mavweb can show evaluation history
|
||||||
7. Add `memory_eval` block to `deploy/mavend.json`
|
7. Add `memory_eval` block to `deploy/mavend.json`
|
||||||
8. Test with synthetic store state — verify observations match expected patterns
|
8. Test with synthetic store state — verify observations match expected patterns
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Status 2026-08-01 — foundation shipped (Vikunja #248)
|
||||||
|
|
||||||
|
**Shipped:** `internal/memeval` (not `internal/memory/eval.go` — `internal/store`
|
||||||
|
imports `internal/memory` for the vector backend, so an evaluator that reads
|
||||||
|
`store.Fact` there would close an import cycle). `Evaluator.Evaluate` reads
|
||||||
|
`RecentFacts` / `RecentNotes` / `RecentNudges`, prompts the resident model under
|
||||||
|
a GBNF grammar for at most three `{observation, confidence, suggested_action}`
|
||||||
|
objects, drops anything under `min_confidence`, deduplicates against what earlier
|
||||||
|
evaluations wrote, and records the rest as notes with source `infer:memory-eval`.
|
||||||
|
Driver: `cmd/mavend/memoryeval.go`, its own goroutine on its own ticker. Config:
|
||||||
|
the `memory_eval` block — **absent ⇒ the loop does not run**. Visibility: `/dash`
|
||||||
|
already renders notes with their source, so evaluation output is visible with no
|
||||||
|
UI change.
|
||||||
|
|
||||||
|
**Deliberately not shipped — this is policy, not an unfinished edge:**
|
||||||
|
|
||||||
|
- *Dispatching observations as care nudges (plan step 4).* An hourly LLM loop
|
||||||
|
with permission to speak is a machine for generating interruptions, and the
|
||||||
|
content is model-generated text about his own life. The evaluator has no
|
||||||
|
dispatcher reference at all, so it cannot reach a channel by accident. Wiring
|
||||||
|
it to `delivery.Dispatcher` is a separate decision with its own opt-in.
|
||||||
|
- *Acting on `suggested_action`.* It is recorded inside the note text and
|
||||||
|
interpreted by nobody. No reminder, routine or fact is created.
|
||||||
|
- *Writing observation embeddings.* Notes are written with a nil embedding, so
|
||||||
|
they stay out of the RAG recall pool. Feeding generated text back into the pool
|
||||||
|
it came from is how a small model starts citing its own guesses as evidence.
|
||||||
|
|
||||||
|
**Deferred, wants a decision or another capability:**
|
||||||
|
|
||||||
|
- *Plan step 6, the `/eval` IPC method and an evaluation-history view.* `/dash`
|
||||||
|
covers reading the output; a dedicated trace surface is worth building once
|
||||||
|
there is real output to look at, and it should probably show the prompt too.
|
||||||
|
- *`RecentEvents`.* The plan lists it; the evaluator reads facts, notes and
|
||||||
|
nudges. Detected action/object events already drive pattern proposals (#43), and
|
||||||
|
duplicating them here would mostly re-derive that.
|
||||||
|
- *Output quality is unmeasured.* There is no fixture for "did she notice
|
||||||
|
something true". The tests cover the machinery — empty store, confidence floor,
|
||||||
|
dedupe, own-notes exclusion, error handling — not the observations. Until
|
||||||
|
someone reads a week of real output on `/dash`, treat the wording and the
|
||||||
|
`min_confidence` default as unvalidated.
|
||||||
|
|||||||
@@ -7,7 +7,6 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
|
||||||
|
|
||||||
"github.com/kami/maven/internal/ipc"
|
"github.com/kami/maven/internal/ipc"
|
||||||
)
|
)
|
||||||
@@ -359,8 +358,12 @@ func TestGate_IpcServer_ChatAllowedForEnrolledCaller(t *testing.T) {
|
|||||||
|
|
||||||
// recordingAPI — a no-op CoreAPI that counts WriteFact invocations; the auth
|
// recordingAPI — a no-op CoreAPI that counts WriteFact invocations; the auth
|
||||||
// check must reject before reaching it, otherwise the refusal leaks into the
|
// check must reject before reaching it, otherwise the refusal leaks into the
|
||||||
// fake's counts and we fail.
|
// fake's counts and we fail. Embeds ipc.UnimplementedCoreAPI so every method
|
||||||
|
// this test doesn't exercise returns ipc.ErrNotImplemented loudly instead of
|
||||||
|
// being hand-stubbed to a canned value nobody checks.
|
||||||
type recordingAPI struct {
|
type recordingAPI struct {
|
||||||
|
ipc.UnimplementedCoreAPI
|
||||||
|
|
||||||
writes int
|
writes int
|
||||||
chats int
|
chats int
|
||||||
}
|
}
|
||||||
@@ -369,86 +372,7 @@ func (r *recordingAPI) WriteFact(_ context.Context, _ ipc.WriteFactReq) (int64,
|
|||||||
r.writes++
|
r.writes++
|
||||||
return int64(r.writes), nil
|
return int64(r.writes), nil
|
||||||
}
|
}
|
||||||
func (r *recordingAPI) LatestFact(_ context.Context, _ string) (ipc.Fact, error) {
|
|
||||||
return ipc.Fact{}, ipc.ErrNoFact
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) LatestFactBySource(_ context.Context, _, _ string) (ipc.Fact, error) {
|
|
||||||
return ipc.Fact{}, ipc.ErrNoFact
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) Since(_ context.Context, _ string, _ time.Time) (time.Duration, error) {
|
|
||||||
return 0, ipc.ErrNoFact
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) Presence(_ context.Context) (ipc.Presence, error) {
|
|
||||||
return ipc.Presence{}, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) CreateReminder(_ context.Context, _ time.Time, _, _ string) (int64, error) {
|
|
||||||
return 1, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) MarkReminder(_ context.Context, _ int64, _ string) error { return nil }
|
|
||||||
func (r *recordingAPI) ListReminders(_ context.Context, _ int) ([]ipc.Reminder, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) TickTrace(_ context.Context) (ipc.TickTrace, error) {
|
|
||||||
return ipc.TickTrace{}, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) MorningStatus(_ context.Context) ([]ipc.MorningRoutineStatus, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) RecordNudge(_ context.Context, _, _, _ string, _ time.Time) (int64, error) {
|
|
||||||
return 1, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) ResolveNudge(_ context.Context, _ int64, _ string, _ time.Time) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) RecentOutcomes(_ context.Context, _ string, _ int) ([]string, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) RecentFacts(_ context.Context, _ int) ([]ipc.Fact, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) CalendarEvents(_ context.Context, _, _ time.Time) ([]ipc.Fact, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) RecentNudges(_ context.Context, _ int) ([]ipc.Nudge, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) WriteNote(_ context.Context, _ time.Time, _ string, _ []float32, _ string) (int64, error) {
|
|
||||||
return 1, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) QueryNotes(_ context.Context, _ []float32, _ int) ([]ipc.Note, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) RecentNotes(_ context.Context, _ int) ([]ipc.Note, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) ProposeTool(_ context.Context, _, _, _ string, _ time.Time) (bool, error) {
|
|
||||||
return false, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) EnableTool(_ context.Context, _ string, _ []string, _ bool, _ string, _ time.Time) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) DisableTool(_ context.Context, _ string) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) DeleteTool(_ context.Context, _ string) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) LookupTool(_ context.Context, _ string) (ipc.Tool, error) {
|
|
||||||
return ipc.Tool{}, ipc.ErrToolNotFound
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) ListTools(_ context.Context, _ string) ([]ipc.Tool, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (r *recordingAPI) RevertFact(_ context.Context, _ string) (int64, error) {
|
|
||||||
return 0, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) ListProposedRoutines(_ context.Context) ([]ipc.ProposedRoutine, error) {
|
|
||||||
return nil, nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) DismissProposedRoutine(_ context.Context, _ int64) error {
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
func (r *recordingAPI) Chat(_ context.Context, text string) (string, error) {
|
func (r *recordingAPI) Chat(_ context.Context, text string) (string, error) {
|
||||||
r.chats++
|
r.chats++
|
||||||
return "echo: " + text, nil
|
return "echo: " + text, nil
|
||||||
|
|||||||
+167
-1
@@ -140,6 +140,16 @@ type Config struct {
|
|||||||
// item. See internal/morning for the evaluation engine. Empty ⇒ disabled.
|
// item. See internal/morning for the evaluation engine. Empty ⇒ disabled.
|
||||||
MorningRoutines []MorningRoutineConfig `json:"morning_routines,omitempty"`
|
MorningRoutines []MorningRoutineConfig `json:"morning_routines,omitempty"`
|
||||||
|
|
||||||
|
// PatternProposals — whether a routine the digestion tick inferred on its
|
||||||
|
// own may be announced, and how often. nil / absent ⇒ silent detection
|
||||||
|
// only: proposals are written for /routines and never announced. See
|
||||||
|
// PatternProposalConfig.
|
||||||
|
PatternProposals *PatternProposalConfig `json:"pattern_proposals,omitempty"`
|
||||||
|
|
||||||
|
// MemoryEval — background memory evaluation (internal/memeval). nil /
|
||||||
|
// absent ⇒ no evaluation loop at all. See MemoryEvalConfig.
|
||||||
|
MemoryEval *MemoryEvalConfig `json:"memory_eval,omitempty"`
|
||||||
|
|
||||||
// Praxis — the ecosystem attention-state service. When configured, maven
|
// Praxis — the ecosystem attention-state service. When configured, maven
|
||||||
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
|
// calls the Praxis HTTP tools API for attention listing and item lifecycle.
|
||||||
// Maven never touches Praxis's database directly (ecosystem invariant: no
|
// Maven never touches Praxis's database directly (ecosystem invariant: no
|
||||||
@@ -257,18 +267,58 @@ type VoiceConfig struct {
|
|||||||
// Default 0.35 if unset.
|
// Default 0.35 if unset.
|
||||||
RouterThreshold float64 `json:"router_threshold,omitempty"`
|
RouterThreshold float64 `json:"router_threshold,omitempty"`
|
||||||
|
|
||||||
|
// LLMRouter — route with the resident model instead of the embedding
|
||||||
|
// classifier. On by default since Vikunja #320.
|
||||||
|
//
|
||||||
|
// Measured on the held-out fixture (ROUTING-EVAL-31-07-2026.md): 63.2% of
|
||||||
|
// intents right against the classifier's 50.0%, and no route errors. It
|
||||||
|
// costs about 1s per turn instead of 30ms.
|
||||||
|
//
|
||||||
|
// It is safe to leave on. The model can refuse — it answers "unknown" when
|
||||||
|
// it cannot route, and the turn drops to the classifier and its clarify
|
||||||
|
// gate. Any LLM error does the same, so a turn never breaks on the model.
|
||||||
|
// Slot extraction runs on LLM decisions too, so acts get their Fn and
|
||||||
|
// reminders their Time.
|
||||||
|
//
|
||||||
|
// Set it false to go back to the classifier, e.g. on a box with no
|
||||||
|
// llama-server or when 1s a turn is too slow.
|
||||||
|
//
|
||||||
|
// It is a pointer so that "missing from the file" and "explicitly false"
|
||||||
|
// are different things: missing means on, false means off. Read it with
|
||||||
|
// UseLLMRouter(), not directly.
|
||||||
|
LLMRouter *bool `json:"llm_router,omitempty"`
|
||||||
|
|
||||||
// QueryMinScore — the note-recall confidence gate. Top cosine below this
|
// QueryMinScore — the note-recall confidence gate. Top cosine below this
|
||||||
// ⇒ "I don't know" instead of a guess. Tuned for the ONNX embedder (0.55);
|
// ⇒ "I don't know" instead of a guess. Tuned for the ONNX embedder (0.55);
|
||||||
// the HashEmbedder floor scores lexically and may never clear it. 0.55
|
// the HashEmbedder floor scores lexically and may never clear it. 0.55
|
||||||
// default if unset.
|
// default if unset.
|
||||||
QueryMinScore float64 `json:"query_min_score,omitempty"`
|
QueryMinScore float64 `json:"query_min_score,omitempty"`
|
||||||
|
|
||||||
|
// QueryMinMargin — the second half of the recall gate: the top hit must
|
||||||
|
// beat the runner-up by more than this. The absolute score above cannot do
|
||||||
|
// the job on its own, because the e5 embedder puts every cosine in one
|
||||||
|
// narrow high band, so a made-up question scores as high as a real one.
|
||||||
|
// The margin asks whether one note is clearly the best instead.
|
||||||
|
// Negative ⇒ off. 0 ⇒ the default below.
|
||||||
|
QueryMinMargin float64 `json:"query_min_margin,omitempty"`
|
||||||
|
|
||||||
|
// ClarifyMaxAttempts — how many clarifying questions she may ask about one
|
||||||
|
// request before she gives up and says she did not understand. Default 3.
|
||||||
|
ClarifyMaxAttempts int `json:"clarify_max_attempts,omitempty"`
|
||||||
|
|
||||||
// Persona — optional prompt prefix that tunes maven's character. Prepended
|
// Persona — optional prompt prefix that tunes maven's character. Prepended
|
||||||
// to every LLM system prompt (nudge phrasing, note queries, general
|
// to every LLM system prompt (nudge phrasing, note queries, general
|
||||||
// knowledge). Empty string ⇒ current hardcoded persona (feminine-gendered
|
// knowledge). Empty string ⇒ current hardcoded persona (feminine-gendered
|
||||||
// Russian self-reference). Example: "Be formal and answer in English only."
|
// Russian self-reference). Example: "Be formal and answer in English only."
|
||||||
Persona string `json:"persona,omitempty"`
|
Persona string `json:"persona,omitempty"`
|
||||||
|
|
||||||
|
// OwnerName / City — optional facts about the owner, added to the shared
|
||||||
|
// context block (internal/persona). Empty is fine: the block still states
|
||||||
|
// who he is grammatically (a man, addressed as "ты") and the current time.
|
||||||
|
// Nothing about correct behaviour may depend on these being filled in.
|
||||||
|
OwnerName string `json:"owner_name,omitempty"`
|
||||||
|
City string `json:"city,omitempty"`
|
||||||
|
|
||||||
// Weather — the weather provider config. nil ⇒ the daemon wires
|
// Weather — the weather provider config. nil ⇒ the daemon wires
|
||||||
// the stub provider (returns ErrNotConfigured — "погода не настроена").
|
// the stub provider (returns ErrNotConfigured — "погода не настроена").
|
||||||
// Set provider to "open-meteo" to use the keyless Open-Meteo API.
|
// Set provider to "open-meteo" to use the keyless Open-Meteo API.
|
||||||
@@ -312,6 +362,62 @@ type DigestConfig struct {
|
|||||||
SeverityCeiling int `json:"severity_ceiling,omitempty"` // max sev batched
|
SeverityCeiling int `json:"severity_ceiling,omitempty"` // max sev batched
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// PatternProposalConfig — announcement policy for routines the digestion tick
|
||||||
|
// inferred by itself (Vikunja #247, #43).
|
||||||
|
//
|
||||||
|
// Detection is always on and always silent by default: the tick writes a
|
||||||
|
// proposed_routines row and the /routines page shows it. Notify is what turns
|
||||||
|
// "she noticed" into "she said something", and it is OFF unless configured —
|
||||||
|
// Maven is not a nag and not autonomous, so a behaviour that speaks without
|
||||||
|
// being asked has to be switched on deliberately, like weather and telegram.
|
||||||
|
//
|
||||||
|
// When Notify is on, the announcement is still heavily restrained:
|
||||||
|
// - at most one proposal per tick, however many were detected;
|
||||||
|
// - at most one per Cooldown across all pairs (not per pair), so a batch of
|
||||||
|
// freshly-detected patterns cannot turn into a queue of interruptions;
|
||||||
|
// - through the ordinary care-class gate (quiet hours / away / snooze), at
|
||||||
|
// sev1 — the lowest severity there is. A proposal is the least urgent
|
||||||
|
// thing Maven can say.
|
||||||
|
//
|
||||||
|
// A pair is only ever announced once, because it is only ever proposed once:
|
||||||
|
// proposed_routines is UNIQUE(action, object) and the row survives dismissal.
|
||||||
|
type PatternProposalConfig struct {
|
||||||
|
// Notify — announce newly inferred routines. Default false.
|
||||||
|
Notify bool `json:"notify,omitempty"`
|
||||||
|
|
||||||
|
// Cooldown — minimum spacing between two proposal announcements. 0 ⇒
|
||||||
|
// DefaultProposalCooldown (24h).
|
||||||
|
Cooldown Duration `json:"cooldown,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// AnnounceProposals reports whether inferred routines may be announced. Safe
|
||||||
|
// on a nil receiver — an absent config block means silent detection.
|
||||||
|
func (p *PatternProposalConfig) AnnounceProposals() bool {
|
||||||
|
return p != nil && p.Notify
|
||||||
|
}
|
||||||
|
|
||||||
|
// MemoryEvalConfig — the background memory-evaluation loop (Vikunja #248).
|
||||||
|
// Absent ⇒ off, like every other capability that costs something the owner did
|
||||||
|
// not ask for. Each evaluation is a full LLM round-trip on the one resident
|
||||||
|
// model, which is the same model answering him; running it hourly by default
|
||||||
|
// would put a multi-second stall in front of an occasional voice turn for a
|
||||||
|
// feature he may not want.
|
||||||
|
//
|
||||||
|
// The loop only ever writes notes (source infer:memory-eval, visible on
|
||||||
|
// /dash). It cannot speak — see internal/memeval.
|
||||||
|
type MemoryEvalConfig struct {
|
||||||
|
// Interval — how often to evaluate. 0 ⇒ DefaultMemoryEvalInterval.
|
||||||
|
Interval Duration `json:"interval,omitempty"`
|
||||||
|
|
||||||
|
// MaxItems — recent facts / notes / nudges fed into one evaluation.
|
||||||
|
// 0 ⇒ memeval.DefaultMaxItems.
|
||||||
|
MaxItems int `json:"max_items,omitempty"`
|
||||||
|
|
||||||
|
// MinConfidence — observations the model scores below this are dropped.
|
||||||
|
// 0 ⇒ memeval.DefaultMinConfidence.
|
||||||
|
MinConfidence float64 `json:"min_confidence,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
// PhraserConfig — the LLM-backed phraser seam. The daemon spawns llama-server
|
// PhraserConfig — the LLM-backed phraser seam. The daemon spawns llama-server
|
||||||
// as a managed subprocess and sends chat-completion requests to phrase nudge
|
// as a managed subprocess and sends chat-completion requests to phrase nudge
|
||||||
// and reminder messages. nil ⇒ the template-based Stub is used instead.
|
// and reminder messages. nil ⇒ the template-based Stub is used instead.
|
||||||
@@ -329,6 +435,12 @@ type PhraserConfig struct {
|
|||||||
NGpuLayers int `json:"n_gpu_layers,omitempty"`
|
NGpuLayers int `json:"n_gpu_layers,omitempty"`
|
||||||
NCtx int `json:"n_ctx,omitempty"`
|
NCtx int `json:"n_ctx,omitempty"`
|
||||||
Timeout Duration `json:"timeout,omitempty"`
|
Timeout Duration `json:"timeout,omitempty"`
|
||||||
|
|
||||||
|
// LLMNudges — let the model word nudges again. Off by default: nudges are
|
||||||
|
// worded from hand-written Russian templates now (the model broke the
|
||||||
|
// persona and invented units). Chat, query and reminder phrasing always go
|
||||||
|
// through the model regardless. See phraser.Config.LLMNudges.
|
||||||
|
LLMNudges bool `json:"llm_nudges,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
// EmbedderConfig — paths for the ONNX multilingual embedder. The daemon
|
// EmbedderConfig — paths for the ONNX multilingual embedder. The daemon
|
||||||
@@ -390,9 +502,27 @@ const (
|
|||||||
DefaultAutotuneInterval = 10 * time.Minute
|
DefaultAutotuneInterval = 10 * time.Minute
|
||||||
DefaultRouterThreshold = 0.55
|
DefaultRouterThreshold = 0.55
|
||||||
DefaultQueryMinScore = 0.55
|
DefaultQueryMinScore = 0.55
|
||||||
DefaultToolTimeout = 30 * time.Second
|
// Read off the margin sweep in internal/memory/recalleval on the e5
|
||||||
|
// embedder: 0.008 answers 68% of real questions (down from 72%) and cuts
|
||||||
|
// false recall from 5/5 to 1/5. Every larger delta costs real recall
|
||||||
|
// without removing that last one until 0.020, which drops recall to 44%.
|
||||||
|
DefaultQueryMinMargin = 0.008
|
||||||
|
// DefaultClarifyMaxAttempts — see dialogue.DefaultMaxAttempts.
|
||||||
|
DefaultClarifyMaxAttempts = 3
|
||||||
|
DefaultToolTimeout = 30 * time.Second
|
||||||
|
// DefaultLLMRouter — route with the resident model unless told otherwise.
|
||||||
|
DefaultLLMRouter = true
|
||||||
|
|
||||||
DefaultFactEnrichmentInterval = 30 * time.Second
|
DefaultFactEnrichmentInterval = 30 * time.Second
|
||||||
|
|
||||||
|
// DefaultProposalCooldown — one inferred-routine announcement per day at
|
||||||
|
// most. A proposal is never urgent; if two patterns surface in the same
|
||||||
|
// hour, the second one waits, and the /routines page has it either way.
|
||||||
|
DefaultProposalCooldown = 24 * time.Hour
|
||||||
|
|
||||||
|
// DefaultMemoryEvalInterval — the plan's cadence (1h) for the memory
|
||||||
|
// evaluation loop, applied only when the block is present at all.
|
||||||
|
DefaultMemoryEvalInterval = time.Hour
|
||||||
)
|
)
|
||||||
|
|
||||||
// Load reads the JSON config at path and applies defaults. A missing file is
|
// Load reads the JSON config at path and applies defaults. A missing file is
|
||||||
@@ -465,6 +595,18 @@ func (c *Config) applyDefaults() {
|
|||||||
c.Digest.SeverityCeiling = 2
|
c.Digest.SeverityCeiling = 2
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Absent block stays nil (⇒ silent detection). Present-but-partial gets the
|
||||||
|
// cooldown default, so `{"notify": true}` is enough to switch it on.
|
||||||
|
if c.PatternProposals != nil && c.PatternProposals.Cooldown <= 0 {
|
||||||
|
c.PatternProposals.Cooldown = Duration(DefaultProposalCooldown)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Same rule: absent stays nil (⇒ no evaluation loop), present gets defaults
|
||||||
|
// so `{}` is a valid "on with the plan's cadence".
|
||||||
|
if c.MemoryEval != nil && c.MemoryEval.Interval <= 0 {
|
||||||
|
c.MemoryEval.Interval = Duration(DefaultMemoryEvalInterval)
|
||||||
|
}
|
||||||
|
|
||||||
if c.Voice != nil {
|
if c.Voice != nil {
|
||||||
if c.Voice.RouterThreshold <= 0 {
|
if c.Voice.RouterThreshold <= 0 {
|
||||||
c.Voice.RouterThreshold = DefaultRouterThreshold
|
c.Voice.RouterThreshold = DefaultRouterThreshold
|
||||||
@@ -472,9 +614,24 @@ func (c *Config) applyDefaults() {
|
|||||||
if c.Voice.QueryMinScore <= 0 {
|
if c.Voice.QueryMinScore <= 0 {
|
||||||
c.Voice.QueryMinScore = DefaultQueryMinScore
|
c.Voice.QueryMinScore = DefaultQueryMinScore
|
||||||
}
|
}
|
||||||
|
// Unset ⇒ default. Negative is how you turn the margin off on purpose,
|
||||||
|
// so it is clamped to 0 rather than replaced by the default.
|
||||||
|
switch {
|
||||||
|
case c.Voice.QueryMinMargin == 0:
|
||||||
|
c.Voice.QueryMinMargin = DefaultQueryMinMargin
|
||||||
|
case c.Voice.QueryMinMargin < 0:
|
||||||
|
c.Voice.QueryMinMargin = 0
|
||||||
|
}
|
||||||
|
if c.Voice.ClarifyMaxAttempts <= 0 {
|
||||||
|
c.Voice.ClarifyMaxAttempts = DefaultClarifyMaxAttempts
|
||||||
|
}
|
||||||
if c.Voice.ToolTimeout <= 0 {
|
if c.Voice.ToolTimeout <= 0 {
|
||||||
c.Voice.ToolTimeout = Duration(DefaultToolTimeout)
|
c.Voice.ToolTimeout = Duration(DefaultToolTimeout)
|
||||||
}
|
}
|
||||||
|
if c.Voice.LLMRouter == nil {
|
||||||
|
on := DefaultLLMRouter
|
||||||
|
c.Voice.LLMRouter = &on
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// routines: default severity to care-class (1) — the safe floor: a
|
// routines: default severity to care-class (1) — the safe floor: a
|
||||||
@@ -493,6 +650,15 @@ func (c *Config) applyDefaults() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// UseLLMRouter reports whether to route with the resident model. Unset means
|
||||||
|
// on; only an explicit false in the config turns it off.
|
||||||
|
func (v *VoiceConfig) UseLLMRouter() bool {
|
||||||
|
if v == nil || v.LLMRouter == nil {
|
||||||
|
return DefaultLLMRouter
|
||||||
|
}
|
||||||
|
return *v.LLMRouter
|
||||||
|
}
|
||||||
|
|
||||||
func (c *Config) validate() error {
|
func (c *Config) validate() error {
|
||||||
if c.Phraser != nil {
|
if c.Phraser != nil {
|
||||||
if c.Phraser.ModelPath == "" {
|
if c.Phraser.ModelPath == "" {
|
||||||
|
|||||||
@@ -35,6 +35,27 @@ func TestLoadDefaults(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Nudges come from templates unless the config says otherwise.
|
||||||
|
func TestPhraserLLMNudgesDefaultsOff(t *testing.T) {
|
||||||
|
p := writeConfig(t, `{"phraser":{"model_path":"/tmp/m.gguf"}}`)
|
||||||
|
c, err := Load(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if c.Phraser.LLMNudges {
|
||||||
|
t.Error("llm_nudges defaults on; templates must be the default")
|
||||||
|
}
|
||||||
|
|
||||||
|
p = writeConfig(t, `{"phraser":{"model_path":"/tmp/m.gguf","llm_nudges":true}}`)
|
||||||
|
c, err = Load(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if !c.Phraser.LLMNudges {
|
||||||
|
t.Error("llm_nudges:true did not parse")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestLoadDurationsParse(t *testing.T) {
|
func TestLoadDurationsParse(t *testing.T) {
|
||||||
p := writeConfig(t, `{"tick_interval":"90s","repeat_interval":"10m"}`)
|
p := writeConfig(t, `{"tick_interval":"90s","repeat_interval":"10m"}`)
|
||||||
c, err := Load(p)
|
c, err := Load(p)
|
||||||
@@ -171,6 +192,40 @@ func TestWeatherConfigNilOK(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestLLMRouterDefaultsOn(t *testing.T) {
|
||||||
|
p := writeConfig(t, `{"voice":{"enabled":true,"bind":"127.0.0.1:9100"}}`)
|
||||||
|
c, err := Load(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if !c.Voice.UseLLMRouter() {
|
||||||
|
t.Error("voice.llm_router absent should mean on")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Missing and explicitly false must not mean the same thing.
|
||||||
|
func TestLLMRouterExplicitFalseTurnsItOff(t *testing.T) {
|
||||||
|
p := writeConfig(t, `{"voice":{"enabled":true,"bind":"127.0.0.1:9100","llm_router":false}}`)
|
||||||
|
c, err := Load(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if c.Voice.UseLLMRouter() {
|
||||||
|
t.Error("voice.llm_router false should turn it off")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestLLMRouterRead(t *testing.T) {
|
||||||
|
p := writeConfig(t, `{"voice":{"enabled":true,"bind":"127.0.0.1:9100","llm_router":true}}`)
|
||||||
|
c, err := Load(p)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if !c.Voice.UseLLMRouter() {
|
||||||
|
t.Error("voice.llm_router true was not read")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestDurationRoundTrip(t *testing.T) {
|
func TestDurationRoundTrip(t *testing.T) {
|
||||||
d := Duration(15 * time.Minute)
|
d := Duration(15 * time.Minute)
|
||||||
b, err := d.MarshalJSON()
|
b, err := d.MarshalJSON()
|
||||||
@@ -188,3 +243,53 @@ func TestDurationRoundTrip(t *testing.T) {
|
|||||||
t.Errorf("round-trip = %v, want %v", d2, d)
|
t.Errorf("round-trip = %v, want %v", d2, d)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Both new opt-in capabilities follow the same rule: absent block ⇒ nil ⇒ the
|
||||||
|
// behaviour does not exist. Presence is the enable act, so a bare `{}` block is
|
||||||
|
// valid and gets the defaults filled in.
|
||||||
|
func TestOptInBlocksAbsentStayNil(t *testing.T) {
|
||||||
|
c, err := Load(writeConfig(t, `{}`))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if c.PatternProposals != nil {
|
||||||
|
t.Errorf("pattern_proposals absent but got %+v", c.PatternProposals)
|
||||||
|
}
|
||||||
|
if c.PatternProposals.AnnounceProposals() {
|
||||||
|
t.Error("AnnounceProposals() true with no config block")
|
||||||
|
}
|
||||||
|
if c.MemoryEval != nil {
|
||||||
|
t.Errorf("memory_eval absent but got %+v", c.MemoryEval)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestOptInBlocksGetDefaultsWhenPresent(t *testing.T) {
|
||||||
|
c, err := Load(writeConfig(t, `{"pattern_proposals":{"notify":true},"memory_eval":{}}`))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if !c.PatternProposals.AnnounceProposals() {
|
||||||
|
t.Error("notify:true did not enable announcements")
|
||||||
|
}
|
||||||
|
if time.Duration(c.PatternProposals.Cooldown) != DefaultProposalCooldown {
|
||||||
|
t.Errorf("proposal cooldown = %v, want %v", c.PatternProposals.Cooldown, DefaultProposalCooldown)
|
||||||
|
}
|
||||||
|
if time.Duration(c.MemoryEval.Interval) != DefaultMemoryEvalInterval {
|
||||||
|
t.Errorf("memory eval interval = %v, want %v", c.MemoryEval.Interval, DefaultMemoryEvalInterval)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Notify is off even when the block exists — the block is where you tune it,
|
||||||
|
// notify:true is the act that lets her speak.
|
||||||
|
func TestPatternProposalNotifyDefaultsOff(t *testing.T) {
|
||||||
|
c, err := Load(writeConfig(t, `{"pattern_proposals":{"cooldown":"6h"}}`))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if c.PatternProposals.AnnounceProposals() {
|
||||||
|
t.Error("notify defaulted to on")
|
||||||
|
}
|
||||||
|
if time.Duration(c.PatternProposals.Cooldown) != 6*time.Hour {
|
||||||
|
t.Errorf("cooldown = %v, want 6h", c.PatternProposals.Cooldown)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -137,9 +137,10 @@ func NewDispatcher(cfg Config) *Dispatcher {
|
|||||||
// picks for (severity, presence), sends via the matching sink, and records
|
// picks for (severity, presence), sends via the matching sink, and records
|
||||||
// one nudge row per successful send. returns the dispatches (one per channel).
|
// one nudge row per successful send. returns the dispatches (one per channel).
|
||||||
//
|
//
|
||||||
// a Drop channel = no send, no record (the nudge was suppressed by routing,
|
// a Drop channel = no send (the nudge was suppressed by routing, not by a
|
||||||
// not by a failure — "a missed water nudge is noise"). a nil sink = channel
|
// failure — "a missed water nudge is noise"), but it does leave a 'dropped'
|
||||||
// not wired, skip silently. a send error stops the dispatch and returns what
|
// outbox row so the suppression is visible. a nil sink = channel not wired,
|
||||||
|
// skip silently. a send error stops the dispatch and returns what
|
||||||
// got through — the daemon decides whether to retry.
|
// got through — the daemon decides whether to retry.
|
||||||
func (d *Dispatcher) DispatchNudge(ctx context.Context, pn PhrasedNudge, now time.Time) ([]Dispatch, error) {
|
func (d *Dispatcher) DispatchNudge(ctx context.Context, pn PhrasedNudge, now time.Time) ([]Dispatch, error) {
|
||||||
c := pn.Candidate
|
c := pn.Candidate
|
||||||
@@ -148,6 +149,16 @@ func (d *Dispatcher) DispatchNudge(ctx context.Context, pn PhrasedNudge, now tim
|
|||||||
for i := 0; i < len(channels); i++ {
|
for i := 0; i < len(channels); i++ {
|
||||||
ch := channels[i]
|
ch := channels[i]
|
||||||
if ch == ChannelDrop {
|
if ch == ChannelDrop {
|
||||||
|
// the routing table suppressed this nudge on purpose (a care nudge
|
||||||
|
// while you're away is noise). that stays — but it must not be
|
||||||
|
// invisible, or "she dropped it" and "the rule never fired" look
|
||||||
|
// the same afterwards. no nudges row: that table feeds the
|
||||||
|
// ignored_rate signal, and a nudge nobody could see must not
|
||||||
|
// count as ignored.
|
||||||
|
id := d.beginOutbox(ctx, "nudge", c.Rule.Name, 0, ch, pn.Summary, now)
|
||||||
|
d.completeOutbox(ctx, id, store.DeliveryDropped, now)
|
||||||
|
log.Printf("dispatcher: dropped %s (sev%d, presence=%s) — routing table suppressed it",
|
||||||
|
c.Rule.Name, c.Severity, c.State.Presence)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
s := Sendable{
|
s := Sendable{
|
||||||
@@ -160,12 +171,20 @@ func (d *Dispatcher) DispatchNudge(ctx context.Context, pn PhrasedNudge, now tim
|
|||||||
RepeatUntilAck: ch == ChannelTelegram && c.Severity >= loop.Sev4,
|
RepeatUntilAck: ch == ChannelTelegram && c.Severity >= loop.Sev4,
|
||||||
Ts: now,
|
Ts: now,
|
||||||
}
|
}
|
||||||
|
s = minimalForAway(s)
|
||||||
sink := d.sinkFor(ch)
|
sink := d.sinkFor(ch)
|
||||||
if sink == nil {
|
if sink == nil {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
attemptID := d.beginOutbox(ctx, "nudge", c.Rule.Name, 0, ch, messageForChannel(s), now)
|
attemptID := d.beginOutbox(ctx, "nudge", c.Rule.Name, 0, ch, messageForChannel(s), now)
|
||||||
if err := sink.Send(ctx, s); err != nil {
|
if err := safeSend(ctx, sink, s); err != nil {
|
||||||
|
if errors.Is(err, ErrSinkPanicked) {
|
||||||
|
// one broken sink must not eat the other channels for this
|
||||||
|
// nudge (sev4 present is voice + ntfy). the attempt is closed
|
||||||
|
// as failed and we move on.
|
||||||
|
d.completeOutbox(ctx, attemptID, store.DeliveryFailed, now)
|
||||||
|
continue
|
||||||
|
}
|
||||||
if errors.Is(err, ErrVoiceNoSession) {
|
if errors.Is(err, ErrVoiceNoSession) {
|
||||||
// voice was assumed reachable (presence=present) but no live
|
// voice was assumed reachable (presence=present) but no live
|
||||||
// session exists — the presence guess was wrong. reroute through
|
// session exists — the presence guess was wrong. reroute through
|
||||||
@@ -225,12 +244,17 @@ func (d *Dispatcher) DispatchReminder(ctx context.Context, pr PhrasedReminder, n
|
|||||||
Summary: pr.Summary,
|
Summary: pr.Summary,
|
||||||
Ts: now,
|
Ts: now,
|
||||||
}
|
}
|
||||||
|
s = minimalForAway(s)
|
||||||
sink := d.sinkFor(ch)
|
sink := d.sinkFor(ch)
|
||||||
if sink == nil {
|
if sink == nil {
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
attemptID := d.beginOutbox(ctx, "reminder", "", rd.Reminder.ID, ch, messageForChannel(s), now)
|
attemptID := d.beginOutbox(ctx, "reminder", "", rd.Reminder.ID, ch, messageForChannel(s), now)
|
||||||
if err := sink.Send(ctx, s); err != nil {
|
if err := safeSend(ctx, sink, s); err != nil {
|
||||||
|
if errors.Is(err, ErrSinkPanicked) {
|
||||||
|
d.completeOutbox(ctx, attemptID, store.DeliveryFailed, now)
|
||||||
|
continue
|
||||||
|
}
|
||||||
if errors.Is(err, ErrVoiceNoSession) {
|
if errors.Is(err, ErrVoiceNoSession) {
|
||||||
// presence guess was wrong — reroute reminder to the away
|
// presence guess was wrong — reroute reminder to the away
|
||||||
// channel (ntfy). voice is the only present channel, so nothing
|
// channel (ntfy). voice is the only present channel, so nothing
|
||||||
@@ -315,9 +339,13 @@ func (d *Dispatcher) RepeatUnacked(ctx context.Context, keys []string, now time.
|
|||||||
RepeatUntilAck: true,
|
RepeatUntilAck: true,
|
||||||
Ts: now,
|
Ts: now,
|
||||||
}
|
}
|
||||||
|
s = minimalForAway(s)
|
||||||
attemptID := d.beginOutbox(ctx, "nudge", key, 0, ChannelTelegram, messageForChannel(s), now)
|
attemptID := d.beginOutbox(ctx, "nudge", key, 0, ChannelTelegram, messageForChannel(s), now)
|
||||||
if err := d.cfg.Telegram.Send(ctx, s); err != nil {
|
if err := safeSend(ctx, d.cfg.Telegram, s); err != nil {
|
||||||
d.completeOutbox(ctx, attemptID, store.DeliveryFailed, now)
|
d.completeOutbox(ctx, attemptID, store.DeliveryFailed, now)
|
||||||
|
if errors.Is(err, ErrSinkPanicked) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
return out, fmt.Errorf("repeat send telegram %s: %w", key, err)
|
return out, fmt.Errorf("repeat send telegram %s: %w", key, err)
|
||||||
}
|
}
|
||||||
d.completeOutbox(ctx, attemptID, store.DeliverySent, now)
|
d.completeOutbox(ctx, attemptID, store.DeliverySent, now)
|
||||||
@@ -329,6 +357,24 @@ func (d *Dispatcher) RepeatUnacked(ctx context.Context, keys []string, now time.
|
|||||||
return out, nil
|
return out, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ErrSinkPanicked — a sink panicked mid-send. the send did not happen, so the
|
||||||
|
// attempt is recorded failed and never silently retried as if it had.
|
||||||
|
var ErrSinkPanicked = errors.New("delivery: sink panicked mid-send")
|
||||||
|
|
||||||
|
// safeSend calls a sink and turns a panic into an error. without this a
|
||||||
|
// panicking sink unwinds past completeOutbox and leaves the delivery_attempts
|
||||||
|
// row pending forever — reconciliation only runs at daemon startup, and core
|
||||||
|
// is long-lived, so the row would sit there for weeks.
|
||||||
|
func safeSend(ctx context.Context, sink Sink, s Sendable) (err error) {
|
||||||
|
defer func() {
|
||||||
|
if r := recover(); r != nil {
|
||||||
|
log.Printf("dispatcher: PANIC in %s sink (this is a bug, fix the sink): %v", s.Channel, r)
|
||||||
|
err = fmt.Errorf("%w: %s: %v", ErrSinkPanicked, s.Channel, r)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
return sink.Send(ctx, s)
|
||||||
|
}
|
||||||
|
|
||||||
func (d *Dispatcher) sinkFor(ch Channel) Sink {
|
func (d *Dispatcher) sinkFor(ch Channel) Sink {
|
||||||
switch ch {
|
switch ch {
|
||||||
case ChannelVoice:
|
case ChannelVoice:
|
||||||
@@ -342,20 +388,51 @@ func (d *Dispatcher) sinkFor(ch Channel) Sink {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// GenericAwayMessage — what an away channel gets when the phraser gave us no
|
||||||
|
// summary. no gendered forms, so it stays right whoever reads it.
|
||||||
|
const GenericAwayMessage = "что-то требует внимания"
|
||||||
|
|
||||||
|
// isAway — this channel leaves the box, so it only ever gets a minimal body.
|
||||||
|
func isAway(ch Channel) bool {
|
||||||
|
return ch == ChannelNtfy || ch == ChannelTelegram
|
||||||
|
}
|
||||||
|
|
||||||
// messageForChannel — away channels get the minimal summary (no shoulder-surf
|
// messageForChannel — away channels get the minimal summary (no shoulder-surf
|
||||||
// exfil — "disk low on homesrv," not detail); voice gets the full body (local).
|
// exfil — "disk low on homesrv," not detail); voice gets the full body (local).
|
||||||
// a missing summary falls back to body — a terse full message is better than
|
// an empty summary must NOT fall back to the body: the resident model is small
|
||||||
// no message, and the phraser should have produced a summary for away-bound
|
// and drops fields often, and the away path crosses the "never phones home"
|
||||||
// severities. this is the "minimal body" rule from the spec, enforced at the
|
// boundary. so we send a fixed generic line plus the rule name instead. voice
|
||||||
// last mile so a phraser bug can't accidentally exfil via the relay.
|
// is local, so it keeps the full body.
|
||||||
func messageForChannel(s Sendable) string {
|
func messageForChannel(s Sendable) string {
|
||||||
switch s.Channel {
|
if !isAway(s.Channel) {
|
||||||
case ChannelNtfy, ChannelTelegram:
|
|
||||||
if s.Summary != "" {
|
|
||||||
return s.Summary
|
|
||||||
}
|
|
||||||
return s.Body
|
|
||||||
default:
|
|
||||||
return s.Body
|
return s.Body
|
||||||
}
|
}
|
||||||
|
return AwayMessage(s)
|
||||||
|
}
|
||||||
|
|
||||||
|
// AwayMessage — the only text an off-box channel may ever carry. Exported so
|
||||||
|
// the away sinks share this one rule instead of each inventing a fallback: the
|
||||||
|
// summary if we have one, otherwise a fixed generic line. Never the body.
|
||||||
|
func AwayMessage(s Sendable) string {
|
||||||
|
if s.Summary != "" {
|
||||||
|
return s.Summary
|
||||||
|
}
|
||||||
|
if s.RuleName != "" {
|
||||||
|
return GenericAwayMessage + ": " + s.RuleName
|
||||||
|
}
|
||||||
|
return GenericAwayMessage
|
||||||
|
}
|
||||||
|
|
||||||
|
// minimalForAway — strips detail from a Sendable bound for an away channel
|
||||||
|
// before any sink sees it. the sinks pick Summary themselves too, but this is
|
||||||
|
// where the boundary actually is: a sink added later must not be able to leak
|
||||||
|
// the full body just by reading the wrong field.
|
||||||
|
func minimalForAway(s Sendable) Sendable {
|
||||||
|
if !isAway(s.Channel) {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
msg := messageForChannel(s)
|
||||||
|
s.Body = msg
|
||||||
|
s.Summary = msg
|
||||||
|
return s
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -639,11 +639,11 @@ func TestDispatchRecurringReminderReschedules(t *testing.T) {
|
|||||||
// ----------------------------- durable outbox --------------------------------
|
// ----------------------------- durable outbox --------------------------------
|
||||||
|
|
||||||
type outboxAttempt struct {
|
type outboxAttempt struct {
|
||||||
kind, rule string
|
kind, rule string
|
||||||
reminderID int64
|
reminderID int64
|
||||||
channel, hash string
|
channel, hash string
|
||||||
status string
|
status string
|
||||||
begunAt, doneAt time.Time
|
begunAt, doneAt time.Time
|
||||||
}
|
}
|
||||||
|
|
||||||
// fakeOutbox — an in-memory Outbox that also lets a test simulate a crash
|
// fakeOutbox — an in-memory Outbox that also lets a test simulate a crash
|
||||||
@@ -784,3 +784,217 @@ func TestDispatchNudge_OutboxBeginFailureDoesNotBlockSend(t *testing.T) {
|
|||||||
t.Fatalf("send should still happen despite outbox begin failure: sends=%v out=%v", voice.sends, out)
|
t.Fatalf("send should still happen despite outbox begin failure: sends=%v out=%v", voice.sends, out)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ----------------------- away channels carry no detail -----------------------
|
||||||
|
|
||||||
|
// panicSink lives in durability_test.go — a second agent wrote the same helper
|
||||||
|
// for the same #369 case, so this file just uses that one.
|
||||||
|
|
||||||
|
// TestAwaySendsGenericLineWhenSummaryEmpty — #368. The phraser is a small
|
||||||
|
// model and drops fields often. An empty Summary must NOT put the full body
|
||||||
|
// on a channel that leaves the box; the away sendable gets a fixed generic
|
||||||
|
// line plus the rule name instead.
|
||||||
|
func TestAwaySendsGenericLineWhenSummaryEmpty(t *testing.T) {
|
||||||
|
ntfy := &fakeSink{}
|
||||||
|
rec := &fakeNudgeRecorder{}
|
||||||
|
d := NewDispatcher(Config{Ntfy: ntfy, Nudges: rec})
|
||||||
|
|
||||||
|
body := "disk /dev/sda1 at 96%, 2.1G free, largest offender /var/lib/docker"
|
||||||
|
_, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("disk-low", loop.Sev3, store.Away),
|
||||||
|
Body: body,
|
||||||
|
Summary: "",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ntfy.sends) != 1 {
|
||||||
|
t.Fatalf("want 1 ntfy send, got %d", len(ntfy.sends))
|
||||||
|
}
|
||||||
|
want := GenericAwayMessage + ": disk-low"
|
||||||
|
got := messageForChannel(ntfy.sends[0])
|
||||||
|
if got != want {
|
||||||
|
t.Fatalf("away message: want %q, got %q", want, got)
|
||||||
|
}
|
||||||
|
if ntfy.sends[0].Body == body {
|
||||||
|
t.Fatal("away sendable still carries the full body")
|
||||||
|
}
|
||||||
|
if len(rec.rows) != 1 || rec.rows[0].message != want {
|
||||||
|
t.Fatalf("recorded message: want %q, got %+v", want, rec.rows)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSev4AwaySendableCarriesNoDetail — the rule is enforced at the
|
||||||
|
// dispatcher, not in each sink. A sink added later must not be able to leak
|
||||||
|
// the body just by reading the wrong field, so neither field may hold detail.
|
||||||
|
func TestSev4AwaySendableCarriesNoDetail(t *testing.T) {
|
||||||
|
tg := &fakeSink{}
|
||||||
|
d := NewDispatcher(Config{Telegram: tg, Ack: newFakeAck()})
|
||||||
|
|
||||||
|
body := "backup job failed: rsync exit 23 on /home/kami, see /var/log/backup.log"
|
||||||
|
_, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("backup-failed", loop.Sev4, store.Away),
|
||||||
|
Body: body,
|
||||||
|
Summary: "бэкап не прошёл",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(tg.sends) != 1 {
|
||||||
|
t.Fatalf("want 1 telegram send, got %d", len(tg.sends))
|
||||||
|
}
|
||||||
|
s := tg.sends[0]
|
||||||
|
if s.Body != "бэкап не прошёл" || s.Summary != "бэкап не прошёл" {
|
||||||
|
t.Fatalf("away sendable should hold only the summary, got body=%q summary=%q", s.Body, s.Summary)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAwayKeepsANonEmptySummary — the normal path is untouched.
|
||||||
|
func TestAwayKeepsANonEmptySummary(t *testing.T) {
|
||||||
|
ntfy := &fakeSink{}
|
||||||
|
d := NewDispatcher(Config{Ntfy: ntfy})
|
||||||
|
|
||||||
|
_, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("cert", loop.Sev3, store.Away),
|
||||||
|
Body: "cert for maven.local expires in 3 days, issuer letsencrypt",
|
||||||
|
Summary: "сертификат истекает",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if got := messageForChannel(ntfy.sends[0]); got != "сертификат истекает" {
|
||||||
|
t.Fatalf("want the summary unchanged, got %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestVoiceStillGetsTheFullBody — voice never leaves the box, so it keeps the
|
||||||
|
// full phrased message even when Summary is empty.
|
||||||
|
func TestVoiceStillGetsTheFullBody(t *testing.T) {
|
||||||
|
voice := &fakeSink{}
|
||||||
|
d := NewDispatcher(Config{Voice: voice})
|
||||||
|
|
||||||
|
body := "ты не пил воду четыре часа"
|
||||||
|
_, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("water", loop.Sev1, store.Present),
|
||||||
|
Body: body,
|
||||||
|
Summary: "",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(voice.sends) != 1 || voice.sends[0].Body != body {
|
||||||
|
t.Fatalf("voice should get the full body, got %+v", voice.sends)
|
||||||
|
}
|
||||||
|
if got := messageForChannel(voice.sends[0]); got != body {
|
||||||
|
t.Fatalf("voice message: want %q, got %q", body, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestReminderAwaySendsGenericLineWhenSummaryEmpty — the reminder path crosses
|
||||||
|
// the same boundary and has its own Sendable construction.
|
||||||
|
func TestReminderAwaySendsGenericLineWhenSummaryEmpty(t *testing.T) {
|
||||||
|
ntfy := &fakeSink{}
|
||||||
|
d := NewDispatcher(Config{Ntfy: ntfy})
|
||||||
|
|
||||||
|
_, err := d.DispatchReminder(context.Background(), PhrasedReminder{
|
||||||
|
Decision: loop.ReminderDecision{
|
||||||
|
Reminder: store.Reminder{ID: 5},
|
||||||
|
State: loop.State{Now: refNow(), Presence: store.Away},
|
||||||
|
},
|
||||||
|
Body: "позвонить в клинику по поводу анализов",
|
||||||
|
Summary: "",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ntfy.sends) != 1 {
|
||||||
|
t.Fatalf("want 1 ntfy send, got %d", len(ntfy.sends))
|
||||||
|
}
|
||||||
|
if got := messageForChannel(ntfy.sends[0]); got != GenericAwayMessage {
|
||||||
|
t.Fatalf("away reminder message: want %q, got %q", GenericAwayMessage, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// --------------------------- a panicking sink -------------------------------
|
||||||
|
|
||||||
|
// TestPanicMidSendResolvesTheAttempt — #369. A panic used to unwind past
|
||||||
|
// completeOutbox and leave the row pending forever, because reconciliation
|
||||||
|
// only runs at daemon startup and core is long-lived.
|
||||||
|
func TestPanicMidSendResolvesTheAttempt(t *testing.T) {
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
d := NewDispatcher(Config{Voice: &panicSink{}, Outbox: ob})
|
||||||
|
|
||||||
|
_, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("water", loop.Sev1, store.Present),
|
||||||
|
Body: "body", Summary: "sum",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("a panicking sink must not fail the dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 1 {
|
||||||
|
t.Fatalf("want 1 outbox attempt, got %d", len(ob.attempts))
|
||||||
|
}
|
||||||
|
if ob.attempts[0].status != store.DeliveryFailed {
|
||||||
|
t.Fatalf("want status failed after a panic, got %q", ob.attempts[0].status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPanicInOneSinkStillDeliversTheOther — sev4 present is voice + ntfy. One
|
||||||
|
// broken sink must not eat the other channel for the same nudge.
|
||||||
|
func TestPanicInOneSinkStillDeliversTheOther(t *testing.T) {
|
||||||
|
bad := &panicSink{}
|
||||||
|
ntfy := &fakeSink{}
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
d := NewDispatcher(Config{Voice: bad, Ntfy: ntfy, Outbox: ob})
|
||||||
|
|
||||||
|
out, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("backup-failed", loop.Sev4, store.Present),
|
||||||
|
Body: "long detailed body", Summary: "бэкап не прошёл",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if bad.calls != 1 {
|
||||||
|
t.Fatalf("want the voice sink called once, got %d", bad.calls)
|
||||||
|
}
|
||||||
|
if len(ntfy.sends) != 1 {
|
||||||
|
t.Fatalf("ntfy should still get the nudge, got %d sends", len(ntfy.sends))
|
||||||
|
}
|
||||||
|
if len(out) != 1 || out[0].Sendable.Channel != ChannelNtfy {
|
||||||
|
t.Fatalf("want only the ntfy dispatch reported, got %+v", out)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 2 {
|
||||||
|
t.Fatalf("want 2 outbox attempts, got %d", len(ob.attempts))
|
||||||
|
}
|
||||||
|
if ob.attempts[0].status != store.DeliveryFailed || ob.attempts[1].status != store.DeliverySent {
|
||||||
|
t.Fatalf("want [failed, sent], got %q %q", ob.attempts[0].status, ob.attempts[1].status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestPanicInReminderSinkResolvesTheAttempt — the reminder path has its own
|
||||||
|
// send call, and a panic there must not leave the reminder marked fired.
|
||||||
|
func TestPanicInReminderSinkResolvesTheAttempt(t *testing.T) {
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
rc := &fakeReminderCompleter{}
|
||||||
|
d := NewDispatcher(Config{Voice: &panicSink{}, Reminders: rc, Outbox: ob})
|
||||||
|
|
||||||
|
out, err := d.DispatchReminder(context.Background(), PhrasedReminder{
|
||||||
|
Decision: loop.ReminderDecision{
|
||||||
|
Reminder: store.Reminder{ID: 9},
|
||||||
|
State: loop.State{Now: refNow(), Presence: store.Present},
|
||||||
|
},
|
||||||
|
Body: "звонок", Summary: "звонок",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("a panicking sink must not fail the dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(out) != 0 {
|
||||||
|
t.Fatalf("nothing was delivered, want no dispatches, got %+v", out)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 1 || ob.attempts[0].status != store.DeliveryFailed {
|
||||||
|
t.Fatalf("want 1 attempt with status failed, got %+v", ob.attempts)
|
||||||
|
}
|
||||||
|
if len(rc.marked) != 0 {
|
||||||
|
t.Fatalf("reminder must stay pending after a panic, got %+v", rc.marked)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,230 @@
|
|||||||
|
package delivery
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/loop"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// panicSink — a sink that dies mid-send. Models the ugly case: the process is
|
||||||
|
// still alive, so startup reconciliation will not run, but the attempt row was
|
||||||
|
// already begun.
|
||||||
|
type panicSink struct{ calls int }
|
||||||
|
|
||||||
|
func (p *panicSink) Send(_ context.Context, _ Sendable) error {
|
||||||
|
p.calls++
|
||||||
|
panic("sink exploded mid-send")
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------------------------- voice fallthrough, per severity -------------------
|
||||||
|
|
||||||
|
// TestVoiceNoSessionFallthroughLeavesOutboxTrail — the fallthrough must be
|
||||||
|
// visible in the ledger too: the voice attempt closes as failed and the away
|
||||||
|
// attempt is a separate row, so an operator can see the reroute happened.
|
||||||
|
func TestVoiceNoSessionFallthroughLeavesOutboxTrail(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
sev loop.Severity
|
||||||
|
wantAt []string // channel per outbox attempt, in order
|
||||||
|
wantEnd []string // status per attempt, in order
|
||||||
|
}{
|
||||||
|
{"sev3 falls through to ntfy", loop.Sev3,
|
||||||
|
[]string{"voice", "ntfy"}, []string{store.DeliveryFailed, store.DeliverySent}},
|
||||||
|
{"sev4 falls through to telegram", loop.Sev4,
|
||||||
|
[]string{"voice", "telegram"}, []string{store.DeliveryFailed, store.DeliverySent}},
|
||||||
|
// care severities still don't reach an away channel; since #370 the
|
||||||
|
// drop itself is a visible row instead of nothing.
|
||||||
|
{"sev1 drops instead of falling through", loop.Sev1,
|
||||||
|
[]string{"voice", "drop"}, []string{store.DeliveryFailed, store.DeliveryDropped}},
|
||||||
|
{"sev2 drops instead of falling through", loop.Sev2,
|
||||||
|
[]string{"voice", "drop"}, []string{store.DeliveryFailed, store.DeliveryDropped}},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
voice := &fakeSink{err: ErrVoiceNoSession}
|
||||||
|
ntfy, telegram := &fakeSink{}, &fakeSink{}
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
d := NewDispatcher(Config{
|
||||||
|
Voice: voice, Ntfy: ntfy, Telegram: telegram,
|
||||||
|
Ack: newFakeAck(), Outbox: ob,
|
||||||
|
})
|
||||||
|
|
||||||
|
if _, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("some_rule", c.sev, store.Present),
|
||||||
|
Body: "detail", Summary: "short",
|
||||||
|
}, refNow()); err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != len(c.wantAt) {
|
||||||
|
t.Fatalf("want %d outbox attempts, got %d (%+v)", len(c.wantAt), len(ob.attempts), ob.attempts)
|
||||||
|
}
|
||||||
|
for i, a := range ob.attempts {
|
||||||
|
if a.channel != c.wantAt[i] || a.status != c.wantEnd[i] {
|
||||||
|
t.Fatalf("attempt %d: want %s/%s, got %s/%s", i, c.wantAt[i], c.wantEnd[i], a.channel, a.status)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// care severities must not reach an away channel — that would
|
||||||
|
// defeat the drop rule.
|
||||||
|
if c.sev <= loop.Sev2 && (len(ntfy.sends) != 0 || len(telegram.sends) != 0) {
|
||||||
|
t.Fatalf("care nudge escaped to an away channel: ntfy=%d telegram=%d",
|
||||||
|
len(ntfy.sends), len(telegram.sends))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------------------------- crash between Begin and Complete ------------------
|
||||||
|
|
||||||
|
// openTestStore — a real store on a temp file. The reconciliation promise is a
|
||||||
|
// SQL promise, so a fake would only test the fake.
|
||||||
|
func openTestStore(t *testing.T) *store.Store {
|
||||||
|
t.Helper()
|
||||||
|
st, err := store.Open(context.Background(), filepath.Join(t.TempDir(), "maven.db"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("open store: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = st.Close() })
|
||||||
|
return st
|
||||||
|
}
|
||||||
|
|
||||||
|
// attemptStatus reads one attempt row back. Returns ok=false when the row is
|
||||||
|
// gone, which would itself be a broken promise (a dropped attempt).
|
||||||
|
func attemptStatus(t *testing.T, st *store.Store, id int64) (status string, completed bool, ok bool) {
|
||||||
|
t.Helper()
|
||||||
|
tx, err := st.DB(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("read tx: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = tx.Rollback() }()
|
||||||
|
var completedTS *int64
|
||||||
|
err = tx.QueryRowContext(context.Background(),
|
||||||
|
`SELECT status, completed_ts FROM delivery_attempts WHERE id = ?`, id).Scan(&status, &completedTS)
|
||||||
|
if err != nil {
|
||||||
|
return "", false, false
|
||||||
|
}
|
||||||
|
return status, completedTS != nil, true
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCrashBetweenBeginAndCompleteBecomesUnknown — simulate the crash window:
|
||||||
|
// Begin lands, the process dies before Complete. Startup reconciliation must
|
||||||
|
// turn that row into "unknown" — neither resent nor dropped, because Maven
|
||||||
|
// cannot know whether the message left the box.
|
||||||
|
func TestCrashBetweenBeginAndCompleteBecomesUnknown(t *testing.T) {
|
||||||
|
st := openTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
sink := &fakeSink{}
|
||||||
|
|
||||||
|
// the crash: intent recorded, no completion.
|
||||||
|
id, err := st.BeginDeliveryAttempt(ctx, "nudge", "disk_low", 0, "telegram", "hash", refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("begin: %v", err)
|
||||||
|
}
|
||||||
|
if s, _, ok := attemptStatus(t, st, id); !ok || s != store.DeliveryPending {
|
||||||
|
t.Fatalf("before reconcile: want pending, got %q ok=%v", s, ok)
|
||||||
|
}
|
||||||
|
|
||||||
|
// restart.
|
||||||
|
n, err := st.ReconcileStaleDeliveryAttempts(ctx, refNow().Add(time.Minute))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("reconcile: %v", err)
|
||||||
|
}
|
||||||
|
if n != 1 {
|
||||||
|
t.Fatalf("want 1 row reconciled, got %d", n)
|
||||||
|
}
|
||||||
|
s, completed, ok := attemptStatus(t, st, id)
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("reconciliation dropped the row; the promise is it is never dropped")
|
||||||
|
}
|
||||||
|
if s != store.DeliveryUnknown {
|
||||||
|
t.Fatalf("want status unknown, got %q", s)
|
||||||
|
}
|
||||||
|
if !completed {
|
||||||
|
t.Fatal("reconciled row should carry a completed_ts")
|
||||||
|
}
|
||||||
|
// not resent: reconciliation is bookkeeping only, it must never push.
|
||||||
|
if len(sink.sends) != 0 {
|
||||||
|
t.Fatalf("reconciliation must not resend, got %d sends", len(sink.sends))
|
||||||
|
}
|
||||||
|
|
||||||
|
// idempotent: a second restart must not churn the row again.
|
||||||
|
n2, err := st.ReconcileStaleDeliveryAttempts(ctx, refNow().Add(2*time.Minute))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("reconcile again: %v", err)
|
||||||
|
}
|
||||||
|
if n2 != 0 {
|
||||||
|
t.Fatalf("second reconcile should find nothing, got %d", n2)
|
||||||
|
}
|
||||||
|
if s2, _, _ := attemptStatus(t, st, id); s2 != store.DeliveryUnknown {
|
||||||
|
t.Fatalf("unknown must stay unknown, got %q", s2)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestUnknownIsNeverResolvedToSentOrFailed — the "never guess" half of the
|
||||||
|
// promise: nothing may quietly turn an unknown into a definite outcome.
|
||||||
|
func TestUnknownIsNeverResolvedToSentOrFailed(t *testing.T) {
|
||||||
|
st := openTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
id, err := st.BeginDeliveryAttempt(ctx, "nudge", "disk_low", 0, "telegram", "hash", refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("begin: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := st.ReconcileStaleDeliveryAttempts(ctx, refNow()); err != nil {
|
||||||
|
t.Fatalf("reconcile: %v", err)
|
||||||
|
}
|
||||||
|
// a late Complete from the old in-flight send must not win.
|
||||||
|
if err := st.CompleteDeliveryAttempt(ctx, id, store.DeliverySent, refNow().Add(time.Minute)); err != nil {
|
||||||
|
t.Fatalf("late complete: %v", err)
|
||||||
|
}
|
||||||
|
if s, _, _ := attemptStatus(t, st, id); s != store.DeliveryUnknown {
|
||||||
|
t.Fatalf("late complete overwrote an unknown outcome: %q", s)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ------------------------- the boring failure modes --------------------------
|
||||||
|
|
||||||
|
// TestSendTimeoutResolvesTheAttempt — a send that times out is a definite
|
||||||
|
// failure from Maven's side, so the row must not be left pending.
|
||||||
|
func TestSendTimeoutResolvesTheAttempt(t *testing.T) {
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
cancel() // the deadline already blew
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
d := NewDispatcher(Config{Ntfy: &fakeSink{err: context.DeadlineExceeded}, Outbox: ob})
|
||||||
|
|
||||||
|
if _, err := d.DispatchNudge(ctx, PhrasedNudge{
|
||||||
|
Candidate: candidate("cert_expiring", loop.Sev3, store.Away),
|
||||||
|
Body: "detail", Summary: "short",
|
||||||
|
}, refNow()); err == nil {
|
||||||
|
t.Fatal("want a timeout error to propagate")
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 1 || ob.attempts[0].status != store.DeliveryFailed {
|
||||||
|
t.Fatalf("timed-out send must close the attempt as failed, got %+v", ob.attempts)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestCompleteFailureLeavesRowPendingForReconciliation — if Complete itself
|
||||||
|
// fails, the row stays pending on purpose. That is the correct ambiguous state
|
||||||
|
// and startup reconciliation is what resolves it.
|
||||||
|
func TestCompleteFailureLeavesRowPendingForReconciliation(t *testing.T) {
|
||||||
|
ob := &fakeOutbox{completeErr: errors.New("db busy")}
|
||||||
|
d := NewDispatcher(Config{Ntfy: &fakeSink{}, Outbox: ob})
|
||||||
|
|
||||||
|
if _, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("cert_expiring", loop.Sev3, store.Away),
|
||||||
|
Body: "detail", Summary: "short",
|
||||||
|
}, refNow()); err != nil {
|
||||||
|
t.Fatalf("a failed outbox complete must not fail the dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 1 || ob.attempts[0].status != store.DeliveryPending {
|
||||||
|
t.Fatalf("want the row left pending, got %+v", ob.attempts)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The panic gap this file used to describe as a skipped test is fixed and
|
||||||
|
// asserted for real in dispatcher_test.go:TestPanicMidSendResolvesTheAttempt.
|
||||||
|
// panicSink stays here because both files use it.
|
||||||
@@ -2,10 +2,10 @@
|
|||||||
//
|
//
|
||||||
// ntfy is the away-channel for sev3 (ops soft) nudges, sev4 (ops hard)
|
// ntfy is the away-channel for sev3 (ops soft) nudges, sev4 (ops hard)
|
||||||
// nudges when present (alongside voice), and reminders when away. the
|
// nudges when present (alongside voice), and reminders when away. the
|
||||||
// message body is the Sendable's Summary — the minimal-body rule from the
|
// message body is delivery.AwayMessage — the minimal-body rule from the
|
||||||
// spec ("disk low on homesrv," not detail; no shoulder-surf exfil through
|
// spec ("disk low on homesrv," not detail; no shoulder-surf exfil through
|
||||||
// the relay). voice gets Body; away channels get Summary, enforced at the
|
// the relay). the dispatcher already strips detail off away sendables; the
|
||||||
// sink so a phraser bug can't exfil.
|
// sink uses the same helper so it can't leak the body on its own either.
|
||||||
//
|
//
|
||||||
// ntfy runs locally (docker, 127.0.0.1:8085, deny-all auth). maven publishes
|
// ntfy runs locally (docker, 127.0.0.1:8085, deny-all auth). maven publishes
|
||||||
// with a dedicated user (write-only to maven-* topics) — the credential is a
|
// with a dedicated user (write-only to maven-* topics) — the credential is a
|
||||||
@@ -69,18 +69,14 @@ func New(cfg Config) (*Sink, error) {
|
|||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
// Send publishes one notification to ntfy. the body is the Sendable's Summary
|
// Send publishes one notification to ntfy. the body is the minimal away
|
||||||
// (minimal body); Title is "maven" (consistent sender identity on the lock
|
// message (never the full body); Title is "maven" (consistent sender identity
|
||||||
// screen — the content is in the body). Priority maps from severity/kind so
|
// on the lock screen — the content is in the body). Priority maps from severity/kind so
|
||||||
// the phone client can ring differently for an alarm vs a soft ops nudge.
|
// the phone client can ring differently for an alarm vs a soft ops nudge.
|
||||||
func (s *Sink) Send(ctx context.Context, d delivery.Sendable) error {
|
func (s *Sink) Send(ctx context.Context, d delivery.Sendable) error {
|
||||||
body := d.Summary
|
// never fall back to d.Body: ntfy leaves the box, so an empty summary gets
|
||||||
if body == "" {
|
// a generic line instead of the full detail.
|
||||||
body = d.Body // terse full message beats no message
|
body := delivery.AwayMessage(d)
|
||||||
}
|
|
||||||
if body == "" {
|
|
||||||
return fmt.Errorf("ntfysink: empty message for %s", d.Channel)
|
|
||||||
}
|
|
||||||
|
|
||||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, s.topicURL(), strings.NewReader(body))
|
req, err := http.NewRequestWithContext(ctx, http.MethodPost, s.topicURL(), strings.NewReader(body))
|
||||||
if err != nil {
|
if err != nil {
|
||||||
|
|||||||
@@ -147,9 +147,9 @@ func TestSendBodyIsSummaryNotFullBody(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSendFallsBackToBodyWhenSummaryEmpty(t *testing.T) {
|
func TestSendNeverSendsTheBodyWhenSummaryEmpty(t *testing.T) {
|
||||||
// a terse full message is better than no message; the phraser should
|
// #368: this used to fall back to the full body. ntfy leaves the box, so
|
||||||
// produce a summary for away-bound severities, but don't silently drop.
|
// an empty summary gets a fixed generic line plus the rule name instead.
|
||||||
rs := newRecordingServer(t, 200, "")
|
rs := newRecordingServer(t, 200, "")
|
||||||
srv := httptest.NewServer(rs.handler())
|
srv := httptest.NewServer(rs.handler())
|
||||||
defer srv.Close()
|
defer srv.Close()
|
||||||
@@ -160,12 +160,15 @@ func TestSendFallsBackToBodyWhenSummaryEmpty(t *testing.T) {
|
|||||||
t.Fatalf("Send: %v", err)
|
t.Fatalf("Send: %v", err)
|
||||||
}
|
}
|
||||||
_, _, body, _, _, _ := rs.snapshot()
|
_, _, body, _, _, _ := rs.snapshot()
|
||||||
if body != s.Body {
|
want := delivery.GenericAwayMessage + ": service_down"
|
||||||
t.Fatalf("fallback body: want %q, got %q", s.Body, body)
|
if body != want {
|
||||||
|
t.Fatalf("body: want %q, got %q", want, body)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSendRejectsEmptyMessage(t *testing.T) {
|
func TestSendNeverSendsAnEmptyMessage(t *testing.T) {
|
||||||
|
// with nothing at all to say we still send the generic line — an away
|
||||||
|
// channel can never carry detail, but it also never goes out blank.
|
||||||
rs := newRecordingServer(t, 200, "")
|
rs := newRecordingServer(t, 200, "")
|
||||||
srv := httptest.NewServer(rs.handler())
|
srv := httptest.NewServer(rs.handler())
|
||||||
defer srv.Close()
|
defer srv.Close()
|
||||||
@@ -173,9 +176,13 @@ func TestSendRejectsEmptyMessage(t *testing.T) {
|
|||||||
sink, _ := New(Config{BaseURL: srv.URL, Topic: "maven"})
|
sink, _ := New(Config{BaseURL: srv.URL, Topic: "maven"})
|
||||||
s := nudgeSendable(loop.Sev3, "")
|
s := nudgeSendable(loop.Sev3, "")
|
||||||
s.Body = ""
|
s.Body = ""
|
||||||
err := sink.Send(context.Background(), s)
|
s.RuleName = ""
|
||||||
if err == nil {
|
if err := sink.Send(context.Background(), s); err != nil {
|
||||||
t.Fatal("want error for empty message")
|
t.Fatalf("Send: %v", err)
|
||||||
|
}
|
||||||
|
_, _, body, _, _, _ := rs.snapshot()
|
||||||
|
if body != delivery.GenericAwayMessage {
|
||||||
|
t.Fatalf("body: want %q, got %q", delivery.GenericAwayMessage, body)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,226 @@
|
|||||||
|
package delivery
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/loop"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// This file walks every cell of the DESIGN.md § "Delivery / channel routing"
|
||||||
|
// table, once as the pure table and once through the dispatcher, so a change
|
||||||
|
// to either side has to break a named cell.
|
||||||
|
//
|
||||||
|
// present away
|
||||||
|
// sev1-2 (care) voice drop
|
||||||
|
// sev3 (soft) voice ntfy, once
|
||||||
|
// sev4 (hard) voice + ntfy telegram, repeat til ack
|
||||||
|
|
||||||
|
type tableCell struct {
|
||||||
|
name string
|
||||||
|
sev loop.Severity
|
||||||
|
presence store.Bucket
|
||||||
|
want []Channel
|
||||||
|
}
|
||||||
|
|
||||||
|
func allTableCells() []tableCell {
|
||||||
|
return []tableCell{
|
||||||
|
{"sev1 present", loop.Sev1, store.Present, []Channel{ChannelVoice}},
|
||||||
|
{"sev2 present", loop.Sev2, store.Present, []Channel{ChannelVoice}},
|
||||||
|
{"sev3 present", loop.Sev3, store.Present, []Channel{ChannelVoice}},
|
||||||
|
{"sev4 present", loop.Sev4, store.Present, []Channel{ChannelVoice, ChannelNtfy}},
|
||||||
|
{"sev1 away", loop.Sev1, store.Away, []Channel{ChannelDrop}},
|
||||||
|
{"sev2 away", loop.Sev2, store.Away, []Channel{ChannelDrop}},
|
||||||
|
{"sev3 away", loop.Sev3, store.Away, []Channel{ChannelNtfy}},
|
||||||
|
{"sev4 away", loop.Sev4, store.Away, []Channel{ChannelTelegram}},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func sameChannels(got, want []Channel) bool {
|
||||||
|
if len(got) != len(want) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
for i := range got {
|
||||||
|
if got[i] != want[i] {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestChannelsForEveryTableCell(t *testing.T) {
|
||||||
|
for _, c := range allTableCells() {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
got := ChannelsFor(c.sev, c.presence)
|
||||||
|
if !sameChannels(got, c.want) {
|
||||||
|
t.Fatalf("%s: want %v, got %v", c.name, c.want, got)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestDispatchNudgeEveryTableCell — the same eight cells end to end: exactly
|
||||||
|
// the wanted channels get a send, and every other channel gets none.
|
||||||
|
func TestDispatchNudgeEveryTableCell(t *testing.T) {
|
||||||
|
for _, c := range allTableCells() {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
voice, ntfy, telegram := &fakeSink{}, &fakeSink{}, &fakeSink{}
|
||||||
|
rec := &fakeNudgeRecorder{}
|
||||||
|
d := NewDispatcher(Config{
|
||||||
|
Voice: voice, Ntfy: ntfy, Telegram: telegram,
|
||||||
|
Ack: newFakeAck(), Nudges: rec,
|
||||||
|
})
|
||||||
|
|
||||||
|
out, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("some_rule", c.sev, c.presence),
|
||||||
|
Body: "full detail body",
|
||||||
|
Summary: "short form",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sent := map[Channel]int{
|
||||||
|
ChannelVoice: len(voice.sends),
|
||||||
|
ChannelNtfy: len(ntfy.sends),
|
||||||
|
ChannelTelegram: len(telegram.sends),
|
||||||
|
}
|
||||||
|
for ch, n := range sent {
|
||||||
|
want := 0
|
||||||
|
for _, w := range c.want {
|
||||||
|
if w == ch {
|
||||||
|
want = 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n != want {
|
||||||
|
t.Fatalf("%s: channel %s got %d sends, want %d", c.name, ch, n, want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// one dispatch and one nudge row per real (non-drop) channel.
|
||||||
|
wantDispatches := 0
|
||||||
|
for _, w := range c.want {
|
||||||
|
if w != ChannelDrop {
|
||||||
|
wantDispatches++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(out) != wantDispatches {
|
||||||
|
t.Fatalf("%s: want %d dispatches, got %d", c.name, wantDispatches, len(out))
|
||||||
|
}
|
||||||
|
if len(rec.rows) != wantDispatches {
|
||||||
|
t.Fatalf("%s: want %d nudge rows, got %d", c.name, wantDispatches, len(rec.rows))
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSev3AwayIsNtfyExactlyOnce — "ntfy, once": one send, and nothing on the
|
||||||
|
// sendable asks for a repeat, so the daemon's repeat driver has no reason to
|
||||||
|
// pick it up.
|
||||||
|
func TestSev3AwayIsNtfyExactlyOnce(t *testing.T) {
|
||||||
|
ntfy := &fakeSink{}
|
||||||
|
ack := newFakeAck()
|
||||||
|
d := NewDispatcher(Config{Ntfy: ntfy, Telegram: &fakeSink{}, Ack: ack})
|
||||||
|
|
||||||
|
out, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("cert_expiring", loop.Sev3, store.Away),
|
||||||
|
Body: "cert detail", Summary: "cert expiring",
|
||||||
|
}, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ntfy.sends) != 1 {
|
||||||
|
t.Fatalf("sev3 away: want exactly 1 ntfy send, got %d", len(ntfy.sends))
|
||||||
|
}
|
||||||
|
if out[0].Sendable.RepeatUntilAck {
|
||||||
|
t.Fatalf("sev3 away must not repeat til ack")
|
||||||
|
}
|
||||||
|
if _, ok := ack.lastSent["cert_expiring"]; ok {
|
||||||
|
t.Fatalf("sev3 away must not enter the ack/repeat tracker")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSev4AwayRepeatsUntilAcked — "telegram, repeat til ack": the initial send
|
||||||
|
// arms the ack clock, the repeat driver re-sends while un-acked, and an ack
|
||||||
|
// stops it.
|
||||||
|
func TestSev4AwayRepeatsUntilAcked(t *testing.T) {
|
||||||
|
telegram := &fakeSink{}
|
||||||
|
ack := newFakeAck()
|
||||||
|
d := NewDispatcher(Config{Telegram: telegram, Ack: ack})
|
||||||
|
ctx := context.Background()
|
||||||
|
|
||||||
|
if _, err := d.DispatchNudge(ctx, PhrasedNudge{
|
||||||
|
Candidate: candidate("disk_low", loop.Sev4, store.Away),
|
||||||
|
Body: "disk detail", Summary: "disk low on homesrv",
|
||||||
|
}, refNow()); err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// two intervals pass, still un-acked → two more sends.
|
||||||
|
for i := 1; i <= 2; i++ {
|
||||||
|
at := refNow().Add(time.Duration(i) * 10 * time.Minute)
|
||||||
|
if _, err := d.RepeatUnacked(ctx, []string{"disk_low"}, at, 5*time.Minute, "disk detail", "disk low on homesrv"); err != nil {
|
||||||
|
t.Fatalf("repeat %d: %v", i, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(telegram.sends) != 3 {
|
||||||
|
t.Fatalf("want 1 initial + 2 repeats = 3 telegram sends, got %d", len(telegram.sends))
|
||||||
|
}
|
||||||
|
|
||||||
|
// acked → no further sends, however long we wait.
|
||||||
|
_ = ack.MarkAcked(ctx, "disk_low")
|
||||||
|
if _, err := d.RepeatUnacked(ctx, []string{"disk_low"}, refNow().Add(time.Hour), 5*time.Minute, "b", "s"); err != nil {
|
||||||
|
t.Fatalf("repeat after ack: %v", err)
|
||||||
|
}
|
||||||
|
if len(telegram.sends) != 3 {
|
||||||
|
t.Fatalf("ack must stop the repeat; got %d sends", len(telegram.sends))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestAwayChannelsGetMinimalBody — what leaves the box is the short form, for
|
||||||
|
// every away cell of the table. messageForChannel is the last-mile choice both
|
||||||
|
// away sinks make too.
|
||||||
|
func TestAwayChannelsGetMinimalBody(t *testing.T) {
|
||||||
|
detail := "disk /mnt/hdd1 on homesrv at 97% — 12GB free, biggest offender /var/lib/docker"
|
||||||
|
short := "disk low on homesrv"
|
||||||
|
|
||||||
|
for _, ch := range []Channel{ChannelNtfy, ChannelTelegram} {
|
||||||
|
t.Run(string(ch), func(t *testing.T) {
|
||||||
|
msg := messageForChannel(Sendable{Channel: ch, Body: detail, Summary: short})
|
||||||
|
if msg != short {
|
||||||
|
t.Fatalf("%s message: want %q, got %q", ch, short, msg)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
if got := messageForChannel(Sendable{Channel: ChannelVoice, Body: detail, Summary: short}); got != detail {
|
||||||
|
t.Fatalf("voice is local and gets the full body, got %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Both away-detail gaps this file used to describe as skipped tests are now
|
||||||
|
// fixed and asserted for real in dispatcher_test.go:
|
||||||
|
// TestSev4AwaySendableCarriesNoDetail and TestAwaySendsGenericLineWhenSummaryEmpty.
|
||||||
|
// An empty Summary no longer means "send the whole body" — it means a short
|
||||||
|
// generic line — so the old expectation here was wrong as well as duplicated.
|
||||||
|
|
||||||
|
// TestCareAwayDropIsRecorded — DESIGN.md's drop is a decision ("a missed water
|
||||||
|
// nudge is noise, a missed backup failure isn't"), so it should be visible
|
||||||
|
// rather than vanish. Today drop is a bare `continue`: no nudge row, no outbox
|
||||||
|
// attempt, no log — nothing an operator can see afterwards. now it leaves a
|
||||||
|
// 'dropped' outbox row.
|
||||||
|
func TestCareAwayDropIsRecorded(t *testing.T) {
|
||||||
|
ob := &fakeOutbox{}
|
||||||
|
d := NewDispatcher(Config{Voice: &fakeSink{}, Nudges: &fakeNudgeRecorder{}, Outbox: ob})
|
||||||
|
|
||||||
|
if _, err := d.DispatchNudge(context.Background(), PhrasedNudge{
|
||||||
|
Candidate: candidate("water", loop.Sev1, store.Away),
|
||||||
|
Body: "drink water", Summary: "water",
|
||||||
|
}, refNow()); err != nil {
|
||||||
|
t.Fatalf("dispatch: %v", err)
|
||||||
|
}
|
||||||
|
if len(ob.attempts) != 1 || ob.attempts[0].channel != string(ChannelDrop) {
|
||||||
|
t.Fatalf("care-away drop should leave a visible record, got %+v", ob.attempts)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -2,11 +2,12 @@
|
|||||||
//
|
//
|
||||||
// telegram is the away-channel for sev4 (ops hard) nudges — "disk-fire alarm
|
// telegram is the away-channel for sev4 (ops hard) nudges — "disk-fire alarm
|
||||||
// at 2am routes to telegram, repeat til ack." the message body is the
|
// at 2am routes to telegram, repeat til ack." the message body is the
|
||||||
// Sendable's Summary — the minimal-body rule from the spec ("disk low on
|
// delivery.AwayMessage — the minimal-body rule from the spec ("disk low on
|
||||||
// homesrv," not detail; no shoulder-surf exfil through the relay). voice gets
|
// homesrv," not detail; no shoulder-surf exfil through the relay). the
|
||||||
// Body; away channels get Summary, enforced at the sink so a phraser bug can't
|
// dispatcher already strips detail off away sendables; the sink uses the same
|
||||||
// exfil. additionally, protect_content=true is passed on every send so the
|
// helper so it can't leak the body on its own either. additionally,
|
||||||
// message can't be forwarded out of the chat — locks the minimal body further.
|
// protect_content=true is passed on every send so the message can't be
|
||||||
|
// forwarded out of the chat — locks the minimal body further.
|
||||||
//
|
//
|
||||||
// telegram's bot API is region-restricted for this homesrv — direct egress to
|
// telegram's bot API is region-restricted for this homesrv — direct egress to
|
||||||
// api.telegram.org is unreliable. the spec's "away channels leave the box —
|
// api.telegram.org is unreliable. the spec's "away channels leave the box —
|
||||||
@@ -140,18 +141,13 @@ type telegramResp struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Send publishes one message to the configured telegram chat. the body is the
|
// Send publishes one message to the configured telegram chat. the body is the
|
||||||
// Sendable's Summary (minimal body); empty Summary falls back to Body (terse
|
// minimal away message (never the full body). protect_content=true so even
|
||||||
// full message beats no message). protect_content=true so a phraser bug (Body
|
// that can't be forwarded onward by the user or a chat observer — locks the
|
||||||
// leaking detail through Summary) can't be forwarded onward by the user or a
|
// minimal-body rule at the channel's own last mile.
|
||||||
// chat observer — locks the minimal-body rule at the channel's own last mile.
|
|
||||||
func (s *Sink) Send(ctx context.Context, d delivery.Sendable) error {
|
func (s *Sink) Send(ctx context.Context, d delivery.Sendable) error {
|
||||||
body := d.Summary
|
// never fall back to d.Body: telegram leaves the box, so an empty summary
|
||||||
if body == "" {
|
// gets a generic line instead of the full detail.
|
||||||
body = d.Body
|
body := delivery.AwayMessage(d)
|
||||||
}
|
|
||||||
if body == "" {
|
|
||||||
return fmt.Errorf("telegramsink: empty message for %s", d.Channel)
|
|
||||||
}
|
|
||||||
|
|
||||||
payload := sendMessageReq{
|
payload := sendMessageReq{
|
||||||
ChatID: s.cfg.ChatID,
|
ChatID: s.cfg.ChatID,
|
||||||
|
|||||||
@@ -173,9 +173,9 @@ func TestSendBodyIsSummaryNotFullBody(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSendFallsBackToBodyWhenSummaryEmpty(t *testing.T) {
|
func TestSendNeverSendsTheBodyWhenSummaryEmpty(t *testing.T) {
|
||||||
// terse full message beats none; the phraser should produce a summary for
|
// #368: this used to fall back to the full body. telegram leaves the box,
|
||||||
// away-bound severities, but don't silently drop.
|
// so an empty summary gets a fixed generic line plus the rule name.
|
||||||
rs := newRecordingServer(t, 200, "")
|
rs := newRecordingServer(t, 200, "")
|
||||||
srv := httptest.NewServer(rs.handler())
|
srv := httptest.NewServer(rs.handler())
|
||||||
defer srv.Close()
|
defer srv.Close()
|
||||||
@@ -188,12 +188,14 @@ func TestSendFallsBackToBodyWhenSummaryEmpty(t *testing.T) {
|
|||||||
_, _, body, _, _ := rs.snapshot()
|
_, _, body, _, _ := rs.snapshot()
|
||||||
var req sendMessageReq
|
var req sendMessageReq
|
||||||
_ = json.Unmarshal([]byte(body), &req)
|
_ = json.Unmarshal([]byte(body), &req)
|
||||||
if req.Text != s.Body {
|
want := delivery.GenericAwayMessage + ": service_down"
|
||||||
t.Fatalf("fallback text: want %q, got %q", s.Body, req.Text)
|
if req.Text != want {
|
||||||
|
t.Fatalf("text: want %q, got %q", want, req.Text)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
func TestSendRejectsEmptyMessage(t *testing.T) {
|
func TestSendNeverSendsAnEmptyMessage(t *testing.T) {
|
||||||
|
// with nothing at all to say we still send the generic line.
|
||||||
rs := newRecordingServer(t, 200, "")
|
rs := newRecordingServer(t, 200, "")
|
||||||
srv := httptest.NewServer(rs.handler())
|
srv := httptest.NewServer(rs.handler())
|
||||||
defer srv.Close()
|
defer srv.Close()
|
||||||
@@ -201,9 +203,15 @@ func TestSendRejectsEmptyMessage(t *testing.T) {
|
|||||||
sink, _ := New(sinkCfg(srv.URL))
|
sink, _ := New(sinkCfg(srv.URL))
|
||||||
s := nudgeSendable(loop.Sev4, "")
|
s := nudgeSendable(loop.Sev4, "")
|
||||||
s.Body = ""
|
s.Body = ""
|
||||||
err := sink.Send(context.Background(), s)
|
s.RuleName = ""
|
||||||
if err == nil {
|
if err := sink.Send(context.Background(), s); err != nil {
|
||||||
t.Fatal("want error for empty message")
|
t.Fatalf("Send: %v", err)
|
||||||
|
}
|
||||||
|
_, _, body, _, _ := rs.snapshot()
|
||||||
|
var req sendMessageReq
|
||||||
|
_ = json.Unmarshal([]byte(body), &req)
|
||||||
|
if req.Text != delivery.GenericAwayMessage {
|
||||||
|
t.Fatalf("text: want %q, got %q", delivery.GenericAwayMessage, req.Text)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,193 @@
|
|||||||
|
package dialogue
|
||||||
|
|
||||||
|
import (
|
||||||
|
"sync"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Slot names one field of Slots. Named type, not a free string, so a missing
|
||||||
|
// slot cannot be misspelled — the question phrasing switches on these.
|
||||||
|
type Slot string
|
||||||
|
|
||||||
|
const (
|
||||||
|
SlotTime Slot = "time" // Slots.Time / HasTime
|
||||||
|
SlotKey Slot = "key" // Slots.Key / HasKey
|
||||||
|
SlotValue Slot = "value" // Slots.Value (paired with Key)
|
||||||
|
SlotFn Slot = "fn" // Slots.Fn / HasFn
|
||||||
|
SlotText Slot = "text" // Slots.Text
|
||||||
|
)
|
||||||
|
|
||||||
|
// DefaultMaxAttempts — how many questions she may ask about one request.
|
||||||
|
// Three, because after three tries the likely problem is that she misheard the
|
||||||
|
// whole request, not one slot — so another question about that slot won't help.
|
||||||
|
// Configurable: voice.clarify_max_attempts.
|
||||||
|
const DefaultMaxAttempts = 3
|
||||||
|
|
||||||
|
// PendingQuestion is what Maven holds while she waits for an answer to an open
|
||||||
|
// question. Unlike the yes/no confirms in cmd/mavend/voice.go, the answer here
|
||||||
|
// is free text that fills a missing slot rather than a verdict.
|
||||||
|
type PendingQuestion struct {
|
||||||
|
Intent Intent // what the router already guessed
|
||||||
|
Slots Slots // what it already filled
|
||||||
|
Missing []Slot // what is still empty, in the order to ask about
|
||||||
|
Utterance string // the user's original raw words
|
||||||
|
Asked time.Time
|
||||||
|
TTL time.Duration
|
||||||
|
Attempts int // questions already asked
|
||||||
|
// MaxAttempts caps Attempts. 0 ⇒ DefaultMaxAttempts.
|
||||||
|
MaxAttempts int
|
||||||
|
}
|
||||||
|
|
||||||
|
// maxAttempts is MaxAttempts with the default filled in.
|
||||||
|
func (q *PendingQuestion) maxAttempts() int {
|
||||||
|
if q.MaxAttempts <= 0 {
|
||||||
|
return DefaultMaxAttempts
|
||||||
|
}
|
||||||
|
return q.MaxAttempts
|
||||||
|
}
|
||||||
|
|
||||||
|
func (q *PendingQuestion) IsExpired(now time.Time) bool {
|
||||||
|
return now.After(q.Asked.Add(q.TTL))
|
||||||
|
}
|
||||||
|
|
||||||
|
// CanAsk reports whether Maven may ask another question about this request.
|
||||||
|
func (q *PendingQuestion) CanAsk() bool {
|
||||||
|
return q.Attempts < q.maxAttempts()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ClarifyStore holds the parked questions. Same shape and locking as
|
||||||
|
// SessionStore: keyed by dialogue id, expired entries dropped on read.
|
||||||
|
type ClarifyStore struct {
|
||||||
|
mu sync.RWMutex
|
||||||
|
questions map[string]*PendingQuestion
|
||||||
|
defaultTTL time.Duration
|
||||||
|
}
|
||||||
|
|
||||||
|
func NewClarifyStore(defaultTTL time.Duration) *ClarifyStore {
|
||||||
|
if defaultTTL <= 0 {
|
||||||
|
// Short, like confirmTTL in voice.go: a clarifying question is a
|
||||||
|
// same-breath gesture, a stale one should not eat a later utterance.
|
||||||
|
defaultTTL = 90 * time.Second
|
||||||
|
}
|
||||||
|
return &ClarifyStore{
|
||||||
|
questions: make(map[string]*PendingQuestion),
|
||||||
|
defaultTTL: defaultTTL,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Put parks a question. Called on a clarify decision (cmd/mavend/clarify.go).
|
||||||
|
func (s *ClarifyStore) Put(id string, q *PendingQuestion) {
|
||||||
|
if q.TTL <= 0 {
|
||||||
|
q.TTL = s.defaultTTL
|
||||||
|
}
|
||||||
|
s.mu.Lock()
|
||||||
|
s.questions[id] = q
|
||||||
|
s.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get returns the live parked question, or nil when there is none.
|
||||||
|
func (s *ClarifyStore) Get(id string, now time.Time) *PendingQuestion {
|
||||||
|
s.mu.RLock()
|
||||||
|
q, ok := s.questions[id]
|
||||||
|
s.mu.RUnlock()
|
||||||
|
if !ok {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if q.IsExpired(now) {
|
||||||
|
s.Delete(id)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return q
|
||||||
|
}
|
||||||
|
|
||||||
|
// TakeExpired reports whether a question was parked here but its TTL ran out,
|
||||||
|
// and drops it. Get drops such a question silently, which leaves the user
|
||||||
|
// thinking his request is still alive — the caller uses this to tell him it is
|
||||||
|
// gone before treating his words as a fresh utterance.
|
||||||
|
func (s *ClarifyStore) TakeExpired(id string, now time.Time) bool {
|
||||||
|
s.mu.Lock()
|
||||||
|
defer s.mu.Unlock()
|
||||||
|
q, ok := s.questions[id]
|
||||||
|
if !ok || !q.IsExpired(now) {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
delete(s.questions, id)
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *ClarifyStore) Delete(id string) {
|
||||||
|
s.mu.Lock()
|
||||||
|
delete(s.questions, id)
|
||||||
|
s.mu.Unlock()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Answer merges the slots parsed from the user's answer into the parked ones.
|
||||||
|
// Only the slots listed in Missing are touched. Within those, a value the answer
|
||||||
|
// carries WINS over what was parked: she asked about this slot, so «нет, в пять»
|
||||||
|
// after «в три» must replace the time, not be thrown away.
|
||||||
|
//
|
||||||
|
// This is the clarify answer only. A correction in a fresh turn ("вообще-то
|
||||||
|
// перенеси на пять") is a different code path (followUpMerge) — not here.
|
||||||
|
//
|
||||||
|
// Parsing the answer text into `answer` is the caller's job; this package must
|
||||||
|
// stay free of internal/router.
|
||||||
|
func (q *PendingQuestion) Answer(text string, answer Slots) Slots {
|
||||||
|
out := q.Slots
|
||||||
|
for _, slot := range q.Missing {
|
||||||
|
switch slot {
|
||||||
|
case SlotTime:
|
||||||
|
if answer.HasTime {
|
||||||
|
out.Time = answer.Time
|
||||||
|
out.HasTime = true
|
||||||
|
}
|
||||||
|
case SlotKey:
|
||||||
|
if answer.HasKey {
|
||||||
|
out.Key = answer.Key
|
||||||
|
out.HasKey = true
|
||||||
|
}
|
||||||
|
case SlotValue:
|
||||||
|
if answer.Value != "" {
|
||||||
|
out.Value = answer.Value
|
||||||
|
}
|
||||||
|
case SlotFn:
|
||||||
|
if answer.HasFn {
|
||||||
|
out.Fn = answer.Fn
|
||||||
|
out.HasFn = true
|
||||||
|
out.Args = append([]string(nil), answer.Args...)
|
||||||
|
}
|
||||||
|
case SlotText:
|
||||||
|
if answer.Text != "" {
|
||||||
|
out.Text = answer.Text
|
||||||
|
} else if out.Text == "" {
|
||||||
|
// No parse for a text slot — the raw answer IS the text.
|
||||||
|
out.Text = text
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// StillMissing lists the slots that are empty in s, out of the ones asked for.
|
||||||
|
// The caller uses it to decide between acting and dropping the request.
|
||||||
|
func StillMissing(want []Slot, s Slots) []Slot {
|
||||||
|
var out []Slot
|
||||||
|
for _, slot := range want {
|
||||||
|
empty := false
|
||||||
|
switch slot {
|
||||||
|
case SlotTime:
|
||||||
|
empty = !s.HasTime
|
||||||
|
case SlotKey:
|
||||||
|
empty = !s.HasKey
|
||||||
|
case SlotValue:
|
||||||
|
empty = s.Value == ""
|
||||||
|
case SlotFn:
|
||||||
|
empty = !s.HasFn
|
||||||
|
case SlotText:
|
||||||
|
empty = s.Text == ""
|
||||||
|
}
|
||||||
|
if empty {
|
||||||
|
out = append(out, slot)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
@@ -0,0 +1,247 @@
|
|||||||
|
package dialogue
|
||||||
|
|
||||||
|
import (
|
||||||
|
"reflect"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
var base = time.Date(2026, 7, 31, 12, 0, 0, 0, time.UTC)
|
||||||
|
|
||||||
|
func TestPendingQuestionIsExpired(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
ttl time.Duration
|
||||||
|
now time.Time
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{"fresh", time.Minute, base.Add(10 * time.Second), false},
|
||||||
|
{"exactly at ttl", time.Minute, base.Add(time.Minute), false},
|
||||||
|
{"past ttl", time.Minute, base.Add(2 * time.Minute), true},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
q := &PendingQuestion{Asked: base, TTL: tc.ttl}
|
||||||
|
if got := q.IsExpired(tc.now); got != tc.want {
|
||||||
|
t.Fatalf("IsExpired = %v, want %v", got, tc.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestClarifyStoreGetPutDelete(t *testing.T) {
|
||||||
|
s := NewClarifyStore(time.Minute)
|
||||||
|
|
||||||
|
if got := s.Get("voice", base); got != nil {
|
||||||
|
t.Fatalf("empty store returned %+v", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
q := &PendingQuestion{Intent: IntentReminder, Missing: []Slot{SlotTime}, Asked: base}
|
||||||
|
s.Put("voice", q)
|
||||||
|
if q.TTL != time.Minute {
|
||||||
|
t.Fatalf("Put did not apply the default TTL, got %v", q.TTL)
|
||||||
|
}
|
||||||
|
if got := s.Get("voice", base.Add(time.Second)); got != q {
|
||||||
|
t.Fatalf("Get returned %+v, want the parked question", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Expired questions are dropped on read, not returned.
|
||||||
|
if got := s.Get("voice", base.Add(2*time.Minute)); got != nil {
|
||||||
|
t.Fatalf("expired Get returned %+v", got)
|
||||||
|
}
|
||||||
|
if got := s.Get("voice", base); got != nil {
|
||||||
|
t.Fatalf("expired question was not deleted: %+v", got)
|
||||||
|
}
|
||||||
|
|
||||||
|
s.Put("voice", &PendingQuestion{Asked: base, TTL: time.Hour})
|
||||||
|
s.Delete("voice")
|
||||||
|
if got := s.Get("voice", base); got != nil {
|
||||||
|
t.Fatalf("Delete left %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestClarifyStoreTakeExpired — TakeExpired reports (and drops) only a question
|
||||||
|
// whose TTL ran out.
|
||||||
|
func TestClarifyStoreTakeExpired(t *testing.T) {
|
||||||
|
s := NewClarifyStore(time.Minute)
|
||||||
|
if s.TakeExpired("voice", base) {
|
||||||
|
t.Fatal("nothing parked ⇒ nothing expired")
|
||||||
|
}
|
||||||
|
s.Put("voice", &PendingQuestion{Missing: []Slot{SlotTime}, Asked: base, TTL: time.Minute})
|
||||||
|
if s.TakeExpired("voice", base.Add(30*time.Second)) {
|
||||||
|
t.Fatal("a live question must not report as expired")
|
||||||
|
}
|
||||||
|
if s.Get("voice", base.Add(30*time.Second)) == nil {
|
||||||
|
t.Fatal("a live question must survive TakeExpired")
|
||||||
|
}
|
||||||
|
if !s.TakeExpired("voice", base.Add(2*time.Minute)) {
|
||||||
|
t.Fatal("a stale question must report as expired")
|
||||||
|
}
|
||||||
|
if s.TakeExpired("voice", base.Add(2*time.Minute)) {
|
||||||
|
t.Fatal("TakeExpired must drop the question, so the second call is false")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestNewClarifyStoreDefaultTTL(t *testing.T) {
|
||||||
|
s := NewClarifyStore(0)
|
||||||
|
q := &PendingQuestion{Asked: base}
|
||||||
|
s.Put("voice", q)
|
||||||
|
if q.TTL != 90*time.Second {
|
||||||
|
t.Fatalf("TTL = %v, want 90s", q.TTL)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnswerFillsOnlyMissingSlots(t *testing.T) {
|
||||||
|
answerTime := base.Add(3 * time.Hour)
|
||||||
|
other := base.Add(9 * time.Hour)
|
||||||
|
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
parked Slots
|
||||||
|
missing []Slot
|
||||||
|
text string
|
||||||
|
answer Slots
|
||||||
|
want Slots
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "fills the missing time",
|
||||||
|
parked: Slots{Text: "напомни позвонить"},
|
||||||
|
missing: []Slot{SlotTime},
|
||||||
|
text: "в три",
|
||||||
|
answer: Slots{Time: answerTime, HasTime: true},
|
||||||
|
want: Slots{Text: "напомни позвонить", Time: answerTime, HasTime: true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
// He restated it: «нет, в пять». The new value wins.
|
||||||
|
name: "a restated time overwrites the parked one",
|
||||||
|
parked: Slots{Time: other, HasTime: true},
|
||||||
|
missing: []Slot{SlotTime},
|
||||||
|
text: "нет, в три",
|
||||||
|
answer: Slots{Time: answerTime, HasTime: true},
|
||||||
|
want: Slots{Time: answerTime, HasTime: true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "ignores slots that were not missing",
|
||||||
|
parked: Slots{Key: "water", HasKey: true},
|
||||||
|
missing: []Slot{SlotValue},
|
||||||
|
text: "два литра",
|
||||||
|
answer: Slots{Key: "sleep", HasKey: true, Value: "2l"},
|
||||||
|
want: Slots{Key: "water", HasKey: true, Value: "2l"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "fills key when empty",
|
||||||
|
parked: Slots{},
|
||||||
|
missing: []Slot{SlotKey, SlotValue},
|
||||||
|
text: "воды",
|
||||||
|
answer: Slots{Key: "water", HasKey: true, Value: `"drank"`},
|
||||||
|
want: Slots{Key: "water", HasKey: true, Value: `"drank"`},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "fills fn and its args",
|
||||||
|
parked: Slots{},
|
||||||
|
missing: []Slot{SlotFn},
|
||||||
|
text: "перезапусти nginx",
|
||||||
|
answer: Slots{Fn: "restart", Args: []string{"nginx"}, HasFn: true},
|
||||||
|
want: Slots{Fn: "restart", Args: []string{"nginx"}, HasFn: true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "a restated fn replaces the fn and its args",
|
||||||
|
parked: Slots{Fn: "restart", Args: []string{"nginx"}, HasFn: true},
|
||||||
|
missing: []Slot{SlotFn},
|
||||||
|
text: "останови postgres",
|
||||||
|
answer: Slots{Fn: "stop", Args: []string{"postgres"}, HasFn: true},
|
||||||
|
want: Slots{Fn: "stop", Args: []string{"postgres"}, HasFn: true},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "raw answer becomes the text when nothing was parsed",
|
||||||
|
parked: Slots{},
|
||||||
|
missing: []Slot{SlotText},
|
||||||
|
text: "купить хлеб",
|
||||||
|
answer: Slots{},
|
||||||
|
want: Slots{Text: "купить хлеб"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "parsed text wins over the raw answer",
|
||||||
|
parked: Slots{},
|
||||||
|
missing: []Slot{SlotText},
|
||||||
|
text: "запиши купить хлеб",
|
||||||
|
answer: Slots{Text: "купить хлеб"},
|
||||||
|
want: Slots{Text: "купить хлеб"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty answer leaves the slot missing",
|
||||||
|
parked: Slots{Text: "напомни"},
|
||||||
|
missing: []Slot{SlotTime},
|
||||||
|
text: "не знаю",
|
||||||
|
answer: Slots{},
|
||||||
|
want: Slots{Text: "напомни"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
q := &PendingQuestion{Slots: tc.parked, Missing: tc.missing, Asked: base}
|
||||||
|
// Whole-struct compare: a new field in Slots is covered for free.
|
||||||
|
if got := q.Answer(tc.text, tc.answer); !reflect.DeepEqual(got, tc.want) {
|
||||||
|
t.Fatalf("Answer = %+v, want %+v", got, tc.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCanAskAllowsThreeQuestionsByDefault(t *testing.T) {
|
||||||
|
if DefaultMaxAttempts != 3 {
|
||||||
|
t.Fatalf("DefaultMaxAttempts = %d, want 3", DefaultMaxAttempts)
|
||||||
|
}
|
||||||
|
q := &PendingQuestion{Asked: base} // MaxAttempts unset ⇒ the default
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
if !q.CanAsk() {
|
||||||
|
t.Fatalf("question %d should be allowed", i+1)
|
||||||
|
}
|
||||||
|
q.Attempts++
|
||||||
|
}
|
||||||
|
if q.CanAsk() {
|
||||||
|
t.Fatal("a fourth question must not be allowed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCanAskHonoursConfiguredMax(t *testing.T) {
|
||||||
|
q := &PendingQuestion{Asked: base, MaxAttempts: 1, Attempts: 1}
|
||||||
|
if q.CanAsk() {
|
||||||
|
t.Fatal("MaxAttempts 1 means one question only")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestStillMissing(t *testing.T) {
|
||||||
|
want := []Slot{SlotTime, SlotKey, SlotValue, SlotFn, SlotText}
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
slots Slots
|
||||||
|
want []Slot
|
||||||
|
}{
|
||||||
|
{"all empty", Slots{}, want},
|
||||||
|
{
|
||||||
|
name: "all filled",
|
||||||
|
slots: Slots{Time: base, HasTime: true, Key: "water", HasKey: true, Value: "1l", Fn: "restart", HasFn: true, Text: "t"},
|
||||||
|
want: nil,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "only value left",
|
||||||
|
slots: Slots{Time: base, HasTime: true, Key: "water", HasKey: true, Fn: "restart", HasFn: true, Text: "t"},
|
||||||
|
want: []Slot{SlotValue},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
got := StillMissing(want, tc.slots)
|
||||||
|
if len(got) != len(tc.want) {
|
||||||
|
t.Fatalf("StillMissing = %v, want %v", got, tc.want)
|
||||||
|
}
|
||||||
|
for i := range got {
|
||||||
|
if got[i] != tc.want[i] {
|
||||||
|
t.Fatalf("StillMissing = %v, want %v", got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,8 +1,12 @@
|
|||||||
package dialogue
|
package dialogue
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
"sync"
|
"sync"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
)
|
)
|
||||||
|
|
||||||
type Intent string
|
type Intent string
|
||||||
@@ -21,6 +25,7 @@ type Slots struct {
|
|||||||
Time time.Time
|
Time time.Time
|
||||||
HasTime bool
|
HasTime bool
|
||||||
Key string
|
Key string
|
||||||
|
Value string // payload for a fact key, mirrors router.Slots.Value
|
||||||
HasKey bool
|
HasKey bool
|
||||||
Text string
|
Text string
|
||||||
Fn string
|
Fn string
|
||||||
@@ -49,10 +54,22 @@ func (s *Session) IsExpired(now time.Time) bool {
|
|||||||
return now.After(s.Timestamp.Add(s.TTL))
|
return now.After(s.Timestamp.Add(s.TTL))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// SessionPersister — the bit of the store the session needs, as an interface
|
||||||
|
// so tests can swap it out. Data is an opaque blob: the store never looks
|
||||||
|
// inside, we encode the session as JSON here.
|
||||||
|
type SessionPersister interface {
|
||||||
|
SaveDialogueSession(ctx context.Context, id string, data []byte, ts time.Time, ttl time.Duration) error
|
||||||
|
DeleteDialogueSession(ctx context.Context, id string) error
|
||||||
|
LoadDialogueSessions(ctx context.Context, now time.Time) ([]store.DialogueSessionRow, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// SessionStore keeps the live sessions in a map (the fast path) and mirrors
|
||||||
|
// every write to the persister, so a daemon restart can load them back.
|
||||||
type SessionStore struct {
|
type SessionStore struct {
|
||||||
mu sync.RWMutex
|
mu sync.RWMutex
|
||||||
sessions map[string]*Session
|
sessions map[string]*Session
|
||||||
defaultTTL time.Duration
|
defaultTTL time.Duration
|
||||||
|
persist SessionPersister // may be nil: memory only (tests, no-store paths)
|
||||||
}
|
}
|
||||||
|
|
||||||
func NewSessionStore(defaultTTL time.Duration) *SessionStore {
|
func NewSessionStore(defaultTTL time.Duration) *SessionStore {
|
||||||
@@ -65,6 +82,43 @@ func NewSessionStore(defaultTTL time.Duration) *SessionStore {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// NewPersistentSessionStore — same store, but writes also go to the DB.
|
||||||
|
// Call Load once after this to bring back sessions from a previous run.
|
||||||
|
func NewPersistentSessionStore(defaultTTL time.Duration, p SessionPersister) *SessionStore {
|
||||||
|
s := NewSessionStore(defaultTTL)
|
||||||
|
s.persist = p
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// Load — read the saved sessions back into memory. Anything past its TTL is
|
||||||
|
// dropped (and deleted from the DB by the store), never revived.
|
||||||
|
func (s *SessionStore) Load(ctx context.Context, now time.Time) error {
|
||||||
|
if s.persist == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
rows, err := s.persist.LoadDialogueSessions(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
s.mu.Lock()
|
||||||
|
defer s.mu.Unlock()
|
||||||
|
for _, r := range rows {
|
||||||
|
var sess Session
|
||||||
|
if err := json.Unmarshal(r.Data, &sess); err != nil {
|
||||||
|
// A blob we can't read is not worth failing a startup over.
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if sess.TTL <= 0 {
|
||||||
|
sess.TTL = r.TTL
|
||||||
|
}
|
||||||
|
if sess.IsExpired(now) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
s.sessions[r.ID] = &sess
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
func (s *SessionStore) Get(id string, now time.Time) *Session {
|
func (s *SessionStore) Get(id string, now time.Time) *Session {
|
||||||
s.mu.RLock()
|
s.mu.RLock()
|
||||||
sess, ok := s.sessions[id]
|
sess, ok := s.sessions[id]
|
||||||
@@ -86,12 +140,33 @@ func (s *SessionStore) Put(id string, sess *Session) {
|
|||||||
s.mu.Lock()
|
s.mu.Lock()
|
||||||
s.sessions[id] = sess
|
s.sessions[id] = sess
|
||||||
s.mu.Unlock()
|
s.mu.Unlock()
|
||||||
|
s.save(id, sess)
|
||||||
}
|
}
|
||||||
|
|
||||||
func (s *SessionStore) Delete(id string) {
|
func (s *SessionStore) Delete(id string) {
|
||||||
s.mu.Lock()
|
s.mu.Lock()
|
||||||
delete(s.sessions, id)
|
delete(s.sessions, id)
|
||||||
s.mu.Unlock()
|
s.mu.Unlock()
|
||||||
|
if s.persist != nil {
|
||||||
|
_ = s.persist.DeleteDialogueSession(context.Background(), id)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// save — mirror one session to the DB. Best effort: memory already has it, so
|
||||||
|
// a write error costs us the restart safety net, not the current turn.
|
||||||
|
func (s *SessionStore) save(id string, sess *Session) {
|
||||||
|
if s.persist == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
data, err := json.Marshal(sess)
|
||||||
|
if err != nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
ts := sess.Timestamp
|
||||||
|
if ts.IsZero() {
|
||||||
|
ts = time.Now()
|
||||||
|
}
|
||||||
|
_ = s.persist.SaveDialogueSession(context.Background(), id, data, ts, sess.TTL)
|
||||||
}
|
}
|
||||||
|
|
||||||
func InheritSlots(prev, cur Slots) Slots {
|
func InheritSlots(prev, cur Slots) Slots {
|
||||||
@@ -104,6 +179,9 @@ func InheritSlots(prev, cur Slots) Slots {
|
|||||||
out.Key = prev.Key
|
out.Key = prev.Key
|
||||||
out.HasKey = true
|
out.HasKey = true
|
||||||
}
|
}
|
||||||
|
if out.Value == "" && prev.Value != "" {
|
||||||
|
out.Value = prev.Value
|
||||||
|
}
|
||||||
if out.Text == "" && prev.Text != "" {
|
if out.Text == "" && prev.Text != "" {
|
||||||
out.Text = prev.Text
|
out.Text = prev.Text
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,118 @@
|
|||||||
|
package dialogue
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"path/filepath"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// openStore — a store on disk, so a second handle can reopen the same file.
|
||||||
|
func openStore(t *testing.T, path string) *store.Store {
|
||||||
|
t.Helper()
|
||||||
|
s, err := store.Open(context.Background(), path)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("open store: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = s.Close() })
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// A session written before a restart comes back and still merges a follow-up.
|
||||||
|
func TestSessionSurvivesRestart(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
path := filepath.Join(t.TempDir(), "maven_test.db")
|
||||||
|
now := time.Now().UTC().Truncate(time.Millisecond)
|
||||||
|
|
||||||
|
first := openStore(t, path)
|
||||||
|
before := NewPersistentSessionStore(2*time.Minute, first)
|
||||||
|
before.Put("voice", &Session{
|
||||||
|
Intent: IntentReminder,
|
||||||
|
Slots: Slots{Text: "полить цветы", Time: now.Add(time.Hour), HasTime: true},
|
||||||
|
Timestamp: now,
|
||||||
|
TTL: 2 * time.Minute,
|
||||||
|
})
|
||||||
|
if err := first.Close(); err != nil {
|
||||||
|
t.Fatalf("close: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// fresh handle, fresh in-memory map — as after a daemon restart
|
||||||
|
after := openStore(t, path)
|
||||||
|
reloaded := NewPersistentSessionStore(2*time.Minute, after)
|
||||||
|
if err := reloaded.Load(ctx, now.Add(10*time.Second)); err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
sess := reloaded.Get("voice", now.Add(10*time.Second))
|
||||||
|
if sess == nil {
|
||||||
|
t.Fatal("session did not survive the restart")
|
||||||
|
}
|
||||||
|
if sess.Intent != IntentReminder {
|
||||||
|
t.Fatalf("intent = %q, want reminder", sess.Intent)
|
||||||
|
}
|
||||||
|
// the follow-up carries no text of its own; it must inherit the old one
|
||||||
|
merged := InheritSlots(sess.Slots, Slots{Time: now.Add(2 * time.Hour), HasTime: true})
|
||||||
|
if merged.Text != "полить цветы" {
|
||||||
|
t.Fatalf("merged text = %q, want the earlier turn's text", merged.Text)
|
||||||
|
}
|
||||||
|
if !merged.Time.Equal(now.Add(2 * time.Hour)) {
|
||||||
|
t.Fatalf("merged time = %v, want the follow-up's time", merged.Time)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A session past its TTL is dead: a restart must not bring it back.
|
||||||
|
func TestExpiredSessionNotResurrected(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
path := filepath.Join(t.TempDir(), "maven_test.db")
|
||||||
|
now := time.Now().UTC().Truncate(time.Millisecond)
|
||||||
|
|
||||||
|
first := openStore(t, path)
|
||||||
|
before := NewPersistentSessionStore(2*time.Minute, first)
|
||||||
|
before.Put("voice", &Session{
|
||||||
|
Intent: IntentReminder,
|
||||||
|
Slots: Slots{Text: "полить цветы"},
|
||||||
|
Timestamp: now,
|
||||||
|
TTL: time.Minute,
|
||||||
|
})
|
||||||
|
if err := first.Close(); err != nil {
|
||||||
|
t.Fatalf("close: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
after := openStore(t, path)
|
||||||
|
reloaded := NewPersistentSessionStore(2*time.Minute, after)
|
||||||
|
later := now.Add(5 * time.Minute) // well past the 1-min TTL
|
||||||
|
if err := reloaded.Load(ctx, later); err != nil {
|
||||||
|
t.Fatalf("Load: %v", err)
|
||||||
|
}
|
||||||
|
if sess := reloaded.Get("voice", later); sess != nil {
|
||||||
|
t.Fatalf("expired session came back: %+v", sess)
|
||||||
|
}
|
||||||
|
// and it is gone from the DB too, not just from memory
|
||||||
|
rows, err := after.LoadDialogueSessions(ctx, later)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("LoadDialogueSessions: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 0 {
|
||||||
|
t.Fatalf("expired row still in the DB: %+v", rows)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Delete removes the row as well, so an ended conversation stays ended.
|
||||||
|
func TestDeleteRemovesPersistedSession(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
path := filepath.Join(t.TempDir(), "maven_test.db")
|
||||||
|
now := time.Now().UTC().Truncate(time.Millisecond)
|
||||||
|
|
||||||
|
s := openStore(t, path)
|
||||||
|
ss := NewPersistentSessionStore(2*time.Minute, s)
|
||||||
|
ss.Put("voice", &Session{Intent: IntentChat, Timestamp: now, TTL: time.Minute})
|
||||||
|
ss.Delete("voice")
|
||||||
|
rows, err := s.LoadDialogueSessions(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("LoadDialogueSessions: %v", err)
|
||||||
|
}
|
||||||
|
if len(rows) != 0 {
|
||||||
|
t.Fatalf("row survived Delete: %+v", rows)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -113,4 +113,14 @@ func TestInheritSlots(t *testing.T) {
|
|||||||
if inherited6.Text != "какая погода в москве" {
|
if inherited6.Text != "какая погода в москве" {
|
||||||
t.Error("should inherit text when current is empty")
|
t.Error("should inherit text when current is empty")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
prevValue := Slots{Key: "water", HasKey: true, Value: `"drank"`}
|
||||||
|
inherited7 := InheritSlots(prevValue, Slots{})
|
||||||
|
if inherited7.Value != `"drank"` {
|
||||||
|
t.Error("should inherit value when current is empty")
|
||||||
|
}
|
||||||
|
kept := InheritSlots(prevValue, Slots{Value: "2l"})
|
||||||
|
if kept.Value != "2l" {
|
||||||
|
t.Error("should keep current value")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -237,6 +237,10 @@ type dismissProposedRoutineReq struct {
|
|||||||
ID int64 `json:"id"`
|
ID int64 `json:"id"`
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type acceptProposedRoutineReq struct {
|
||||||
|
ID int64 `json:"id"`
|
||||||
|
}
|
||||||
|
|
||||||
// CoreAPI — what core exposes to modules. One Go interface, satisfied by:
|
// CoreAPI — what core exposes to modules. One Go interface, satisfied by:
|
||||||
// - the in-process store adapter (server.go storeAPI) — used by the daemon
|
// - the in-process store adapter (server.go storeAPI) — used by the daemon
|
||||||
// for modules that live in-process for now (router, delivery) and by tests,
|
// for modules that live in-process for now (router, delivery) and by tests,
|
||||||
@@ -286,6 +290,9 @@ type CoreAPI interface {
|
|||||||
ListProposedRoutines(ctx context.Context) ([]ProposedRoutine, error)
|
ListProposedRoutines(ctx context.Context) ([]ProposedRoutine, error)
|
||||||
// DismissProposedRoutine flips a proposed routine to 'dismissed'.
|
// DismissProposedRoutine flips a proposed routine to 'dismissed'.
|
||||||
DismissProposedRoutine(ctx context.Context, id int64) error
|
DismissProposedRoutine(ctx context.Context, id int64) error
|
||||||
|
// AcceptProposedRoutine flips a proposed routine to 'accepted'. The tick
|
||||||
|
// loop takes the schedule from there — no reminder is created (Vikunja #366).
|
||||||
|
AcceptProposedRoutine(ctx context.Context, id int64) error
|
||||||
|
|
||||||
// TickTrace returns the most recent tick's rule trace. The daemon caches
|
// TickTrace returns the most recent tick's rule trace. The daemon caches
|
||||||
// this after every tick; the store adapter returns an error (trace is not
|
// this after every tick; the store adapter returns an error (trace is not
|
||||||
|
|||||||
@@ -430,6 +430,10 @@ func (c *Client) DismissProposedRoutine(ctx context.Context, id int64) error {
|
|||||||
return c.call(ctx, MethodDismissProposedRoutine, dismissProposedRoutineReq{ID: id}, nil)
|
return c.call(ctx, MethodDismissProposedRoutine, dismissProposedRoutineReq{ID: id}, nil)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (c *Client) AcceptProposedRoutine(ctx context.Context, id int64) error {
|
||||||
|
return c.call(ctx, MethodAcceptProposedRoutine, acceptProposedRoutineReq{ID: id}, nil)
|
||||||
|
}
|
||||||
|
|
||||||
func (c *Client) Chat(ctx context.Context, text string) (string, error) {
|
func (c *Client) Chat(ctx context.Context, text string) (string, error) {
|
||||||
var r chatResp
|
var r chatResp
|
||||||
if err := c.call(ctx, MethodChat, chatReq{Text: text}, &r); err != nil {
|
if err := c.call(ctx, MethodChat, chatReq{Text: text}, &r); err != nil {
|
||||||
|
|||||||
@@ -410,92 +410,12 @@ func TestChatViaClient(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// chatTestAPI — a minimal CoreAPI that only implements Chat for testing.
|
// chatTestAPI — a minimal CoreAPI that only implements Chat for testing.
|
||||||
type chatTestAPI struct{}
|
// Embeds UnimplementedCoreAPI so every other method fails loudly with
|
||||||
|
// ErrNotImplemented instead of needing 27 hand-written no-op stubs.
|
||||||
|
type chatTestAPI struct {
|
||||||
|
UnimplementedCoreAPI
|
||||||
|
}
|
||||||
|
|
||||||
func (a *chatTestAPI) WriteFact(ctx context.Context, req WriteFactReq) (int64, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) LatestFact(ctx context.Context, key string) (Fact, error) {
|
|
||||||
return Fact{}, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) LatestFactBySource(ctx context.Context, key, source string) (Fact, error) {
|
|
||||||
return Fact{}, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) Since(ctx context.Context, key string, now time.Time) (time.Duration, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) Presence(ctx context.Context) (Presence, error) {
|
|
||||||
return Presence{}, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) CreateReminder(ctx context.Context, fire time.Time, payload, cron string) (int64, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) MarkReminder(ctx context.Context, id int64, status string) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) ListReminders(ctx context.Context, n int) ([]Reminder, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RecordNudge(ctx context.Context, rule, channel, message string, ts time.Time) (int64, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) ResolveNudge(ctx context.Context, id int64, outcome string, ts time.Time) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RecentOutcomes(ctx context.Context, rule string, n int) ([]string, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RecentFacts(ctx context.Context, n int) ([]Fact, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) CalendarEvents(ctx context.Context, from, to time.Time) ([]Fact, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RecentNudges(ctx context.Context, n int) ([]Nudge, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) QueryNotes(ctx context.Context, embedding []float32, k int) ([]Note, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RecentNotes(ctx context.Context, n int) ([]Note, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) ProposeTool(ctx context.Context, name, utterance, scope string, ts time.Time) (bool, error) {
|
|
||||||
return false, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) EnableTool(ctx context.Context, name string, cmd []string, destructive bool, scope string, ts time.Time) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) DisableTool(ctx context.Context, name string) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) DeleteTool(ctx context.Context, name string) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) ListProposedRoutines(ctx context.Context) ([]ProposedRoutine, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) DismissProposedRoutine(ctx context.Context, id int64) error {
|
|
||||||
return ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) LookupTool(ctx context.Context, name string) (Tool, error) {
|
|
||||||
return Tool{}, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) ListTools(ctx context.Context, status string) ([]Tool, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) RevertFact(ctx context.Context, key string) (int64, error) {
|
|
||||||
return 0, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) TickTrace(ctx context.Context) (TickTrace, error) {
|
|
||||||
return TickTrace{}, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) MorningStatus(ctx context.Context) ([]MorningRoutineStatus, error) {
|
|
||||||
return nil, ErrUnknownMethod
|
|
||||||
}
|
|
||||||
func (a *chatTestAPI) Chat(ctx context.Context, text string) (string, error) {
|
func (a *chatTestAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||||
if text == "привет" {
|
if text == "привет" {
|
||||||
return "и тебе привет!", nil
|
return "и тебе привет!", nil
|
||||||
|
|||||||
+245
-299
@@ -253,6 +253,10 @@ func (a *storeAPI) DismissProposedRoutine(ctx context.Context, id int64) error {
|
|||||||
return mapErr(a.s.DismissProposedRoutine(ctx, id))
|
return mapErr(a.s.DismissProposedRoutine(ctx, id))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (a *storeAPI) AcceptProposedRoutine(ctx context.Context, id int64) error {
|
||||||
|
return mapErr(a.s.AcceptProposedRoutine(ctx, id, time.Now().UTC()))
|
||||||
|
}
|
||||||
|
|
||||||
func toTool(t store.Tool) Tool {
|
func toTool(t store.Tool) Tool {
|
||||||
return Tool{
|
return Tool{
|
||||||
Name: t.Name, Scope: t.Scope, Cmd: t.Cmd, Destructive: t.Destructive,
|
Name: t.Name, Scope: t.Scope, Cmd: t.Cmd, Destructive: t.Destructive,
|
||||||
@@ -497,6 +501,239 @@ func (s *Server) safeDispatch(ctx context.Context, req Request) (result json.Raw
|
|||||||
return s.dispatch(ctx, req)
|
return s.dispatch(ctx, req)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// handlerFunc — one table entry's shape: unmarshal req.Params (if it wants
|
||||||
|
// any), call the matching CoreAPI method against the api passed in, marshal
|
||||||
|
// the result. api is a parameter, not a closed-over field, precisely so a
|
||||||
|
// table built once at package init never pins a stale CoreAPI — see the note
|
||||||
|
// on methodTable below about SetAPI.
|
||||||
|
type handlerFunc func(ctx context.Context, api CoreAPI, raw json.RawMessage) (json.RawMessage, error)
|
||||||
|
|
||||||
|
// withParams adapts a (typed params, typed result) CoreAPI call into a
|
||||||
|
// handlerFunc: unmarshal into P, call fn, marshal R. On error the result is
|
||||||
|
// dropped (marshalResult's output is never read when err != nil — see
|
||||||
|
// serveConn) so every entry can uniformly return early on error without
|
||||||
|
// re-deriving what the pre-table per-arm code used to return in that case.
|
||||||
|
func withParams[P any, R any](fn func(ctx context.Context, api CoreAPI, p P) (R, error)) handlerFunc {
|
||||||
|
return func(ctx context.Context, api CoreAPI, raw json.RawMessage) (json.RawMessage, error) {
|
||||||
|
var p P
|
||||||
|
if err := unmarshalParams(raw, &p); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
r, err := fn(ctx, api, p)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return marshalResult(r), nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// withParamsVoid is withParams for the error-only methods (mark/resolve/
|
||||||
|
// enable/disable/...): params in, no result out, wire reply is always null.
|
||||||
|
func withParamsVoid[P any](fn func(ctx context.Context, api CoreAPI, p P) error) handlerFunc {
|
||||||
|
return func(ctx context.Context, api CoreAPI, raw json.RawMessage) (json.RawMessage, error) {
|
||||||
|
var p P
|
||||||
|
if err := unmarshalParams(raw, &p); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return marshalResult(nil), fn(ctx, api, p)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// withoutParams is withParams for the handful of methods that take no
|
||||||
|
// params at all (Presence, TickTrace, MorningStatus, ListProposedRoutines).
|
||||||
|
// It does NOT call unmarshalParams — matching the pre-table arms, which
|
||||||
|
// never touched req.Params for these four methods.
|
||||||
|
func withoutParams[R any](fn func(ctx context.Context, api CoreAPI) (R, error)) handlerFunc {
|
||||||
|
return func(ctx context.Context, api CoreAPI, _ json.RawMessage) (json.RawMessage, error) {
|
||||||
|
r, err := fn(ctx, api)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return marshalResult(r), nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// methodTable — one entry per CoreAPI-backed method. Built once at package
|
||||||
|
// init, not per-Server and not per-dispatch: entries close over nothing but
|
||||||
|
// the CoreAPI method being called, and dispatch passes in the *current*
|
||||||
|
// api (loaded fresh via s.api.Load() every call, same as before the table
|
||||||
|
// existed) as an argument — so SetAPI's runtime swap (the unlock transition)
|
||||||
|
// is still honored on the very next request with no extra plumbing here.
|
||||||
|
//
|
||||||
|
// MethodAssertStepUp, MethodStoreEncryptionKey and MethodUnlock are NOT in
|
||||||
|
// this table: they bypass CoreAPI entirely (s.StepUp / s.WrapKeyFn /
|
||||||
|
// s.UnlockFn), so dispatch special-cases them before consulting the table.
|
||||||
|
var methodTable = map[Method]handlerFunc{
|
||||||
|
MethodWriteFact: withParams(func(ctx context.Context, api CoreAPI, p WriteFactReq) (idResp, error) {
|
||||||
|
id, err := api.WriteFact(ctx, p)
|
||||||
|
return idResp{ID: id}, err
|
||||||
|
}),
|
||||||
|
MethodLatestFact: withParams(func(ctx context.Context, api CoreAPI, p keyReq) (Fact, error) {
|
||||||
|
return api.LatestFact(ctx, p.Key)
|
||||||
|
}),
|
||||||
|
MethodLatestFactBySource: withParams(func(ctx context.Context, api CoreAPI, p keySourceReq) (Fact, error) {
|
||||||
|
return api.LatestFactBySource(ctx, p.Key, p.Source)
|
||||||
|
}),
|
||||||
|
MethodSince: withParams(func(ctx context.Context, api CoreAPI, p sinceReq) (sinceResp, error) {
|
||||||
|
d, err := api.Since(ctx, p.Key, p.Now)
|
||||||
|
return sinceResp{Dur: d}, err
|
||||||
|
}),
|
||||||
|
MethodPresence: withoutParams(func(ctx context.Context, api CoreAPI) (Presence, error) {
|
||||||
|
return api.Presence(ctx)
|
||||||
|
}),
|
||||||
|
MethodCreateReminder: withParams(func(ctx context.Context, api CoreAPI, p createReminderReq) (idResp, error) {
|
||||||
|
id, err := api.CreateReminder(ctx, p.Fire, p.Payload, p.Cron)
|
||||||
|
return idResp{ID: id}, err
|
||||||
|
}),
|
||||||
|
MethodMarkReminder: withParamsVoid(func(ctx context.Context, api CoreAPI, p markReminderReq) error {
|
||||||
|
return api.MarkReminder(ctx, p.ID, p.Status)
|
||||||
|
}),
|
||||||
|
MethodListReminders: withParams(func(ctx context.Context, api CoreAPI, p nReq) ([]Reminder, error) {
|
||||||
|
out, err := api.ListReminders(ctx, p.N)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Reminder{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodRecordNudge: withParams(func(ctx context.Context, api CoreAPI, p recordNudgeReq) (idResp, error) {
|
||||||
|
id, err := api.RecordNudge(ctx, p.Rule, p.Channel, p.Message, p.Ts)
|
||||||
|
return idResp{ID: id}, err
|
||||||
|
}),
|
||||||
|
MethodResolveNudge: withParamsVoid(func(ctx context.Context, api CoreAPI, p resolveNudgeReq) error {
|
||||||
|
return api.ResolveNudge(ctx, p.ID, p.Outcome, p.Ts)
|
||||||
|
}),
|
||||||
|
MethodRecentOutcomes: withParams(func(ctx context.Context, api CoreAPI, p outcomesReq) ([]string, error) {
|
||||||
|
out, err := api.RecentOutcomes(ctx, p.Rule, p.N)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []string{} // stable non-null on the wire
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodRecentFacts: withParams(func(ctx context.Context, api CoreAPI, p nReq) ([]Fact, error) {
|
||||||
|
out, err := api.RecentFacts(ctx, p.N)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Fact{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodCalendarEvents: withParams(func(ctx context.Context, api CoreAPI, p calendarEventsReq) ([]Fact, error) {
|
||||||
|
out, err := api.CalendarEvents(ctx, p.From, p.To)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Fact{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodRecentNudges: withParams(func(ctx context.Context, api CoreAPI, p nReq) ([]Nudge, error) {
|
||||||
|
out, err := api.RecentNudges(ctx, p.N)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Nudge{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodWriteNote: withParams(func(ctx context.Context, api CoreAPI, p writeNoteReq) (idResp, error) {
|
||||||
|
id, err := api.WriteNote(ctx, p.Ts, p.Text, p.Embedding, p.Source)
|
||||||
|
return idResp{ID: id}, err
|
||||||
|
}),
|
||||||
|
MethodQueryNotes: withParams(func(ctx context.Context, api CoreAPI, p queryNotesReq) ([]Note, error) {
|
||||||
|
out, err := api.QueryNotes(ctx, p.Embedding, p.K)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Note{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodRecentNotes: withParams(func(ctx context.Context, api CoreAPI, p nReq) ([]Note, error) {
|
||||||
|
out, err := api.RecentNotes(ctx, p.N)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Note{}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}),
|
||||||
|
MethodProposeTool: withParams(func(ctx context.Context, api CoreAPI, p proposeToolReq) (proposeToolResp, error) {
|
||||||
|
ok, err := api.ProposeTool(ctx, p.Name, p.Utterance, p.Scope, p.Ts)
|
||||||
|
return proposeToolResp{Proposed: ok}, err
|
||||||
|
}),
|
||||||
|
MethodEnableTool: withParamsVoid(func(ctx context.Context, api CoreAPI, p enableToolReq) error {
|
||||||
|
return api.EnableTool(ctx, p.Name, p.Cmd, p.Destructive, p.Scope, p.Ts)
|
||||||
|
}),
|
||||||
|
MethodDisableTool: withParamsVoid(func(ctx context.Context, api CoreAPI, p disableToolReq) error {
|
||||||
|
return api.DisableTool(ctx, p.Name)
|
||||||
|
}),
|
||||||
|
MethodLookupTool: withParams(func(ctx context.Context, api CoreAPI, p lookupToolReq) (Tool, error) {
|
||||||
|
return api.LookupTool(ctx, p.Name)
|
||||||
|
}),
|
||||||
|
MethodListTools: withParams(func(ctx context.Context, api CoreAPI, p listToolsReq) (listToolsResp, error) {
|
||||||
|
out, err := api.ListTools(ctx, p.Status)
|
||||||
|
if err != nil {
|
||||||
|
return listToolsResp{}, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []Tool{}
|
||||||
|
}
|
||||||
|
return listToolsResp{Tools: out}, nil
|
||||||
|
}),
|
||||||
|
// MethodDeleteTool shares disableToolReq — both take just a tool name.
|
||||||
|
MethodDeleteTool: withParamsVoid(func(ctx context.Context, api CoreAPI, p disableToolReq) error {
|
||||||
|
return api.DeleteTool(ctx, p.Name)
|
||||||
|
}),
|
||||||
|
MethodListProposedRoutines: withoutParams(func(ctx context.Context, api CoreAPI) (listProposedRoutinesResp, error) {
|
||||||
|
out, err := api.ListProposedRoutines(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return listProposedRoutinesResp{}, err
|
||||||
|
}
|
||||||
|
if out == nil {
|
||||||
|
out = []ProposedRoutine{}
|
||||||
|
}
|
||||||
|
return listProposedRoutinesResp{Routines: out}, nil
|
||||||
|
}),
|
||||||
|
MethodDismissProposedRoutine: withParamsVoid(func(ctx context.Context, api CoreAPI, p dismissProposedRoutineReq) error {
|
||||||
|
return api.DismissProposedRoutine(ctx, p.ID)
|
||||||
|
}),
|
||||||
|
MethodAcceptProposedRoutine: withParamsVoid(func(ctx context.Context, api CoreAPI, p acceptProposedRoutineReq) error {
|
||||||
|
return api.AcceptProposedRoutine(ctx, p.ID)
|
||||||
|
}),
|
||||||
|
MethodRevertFact: withParams(func(ctx context.Context, api CoreAPI, p revertReq) (map[string]int64, error) {
|
||||||
|
newID, err := api.RevertFact(ctx, p.Key)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
return map[string]int64{"new_id": newID}, nil
|
||||||
|
}),
|
||||||
|
MethodChat: withParams(func(ctx context.Context, api CoreAPI, p chatReq) (chatResp, error) {
|
||||||
|
reply, err := api.Chat(ctx, p.Text)
|
||||||
|
return chatResp{Reply: reply}, err
|
||||||
|
}),
|
||||||
|
MethodTickTrace: withoutParams(func(ctx context.Context, api CoreAPI) (TickTrace, error) {
|
||||||
|
return api.TickTrace(ctx)
|
||||||
|
}),
|
||||||
|
// MorningStatus intentionally has no nil→[]T{} normalization here — the
|
||||||
|
// pre-table arm marshaled api.MorningStatus's result as-is (a nil slice
|
||||||
|
// serializes as JSON null), and this preserves that exact wire shape.
|
||||||
|
MethodMorningStatus: withoutParams(func(ctx context.Context, api CoreAPI) ([]MorningRoutineStatus, error) {
|
||||||
|
return api.MorningStatus(ctx)
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
|
||||||
// dispatch unmarshals params for req.Method and calls the matching CoreAPI
|
// dispatch unmarshals params for req.Method and calls the matching CoreAPI
|
||||||
// method. Unknown method ⇒ ErrUnknownMethod; a malformed params payload ⇒
|
// method. Unknown method ⇒ ErrUnknownMethod; a malformed params payload ⇒
|
||||||
// ErrBadParams with the underlying text (local, server-side, not shipped to
|
// ErrBadParams with the underlying text (local, server-side, not shipped to
|
||||||
@@ -513,305 +750,11 @@ func (s *Server) dispatch(ctx context.Context, req Request) (json.RawMessage, er
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// These three bypass CoreAPI entirely — they drive Server fields set
|
||||||
|
// directly by the daemon (StepUp / WrapKeyFn / UnlockFn), not store
|
||||||
|
// state, so they can never be table entries keyed on a CoreAPI method.
|
||||||
switch req.Method {
|
switch req.Method {
|
||||||
case MethodWriteFact:
|
|
||||||
var p WriteFactReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
id, err := api.WriteFact(ctx, p)
|
|
||||||
return marshalResult(idResp{ID: id}), err
|
|
||||||
|
|
||||||
case MethodLatestFact:
|
|
||||||
var p keyReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
f, err := api.LatestFact(ctx, p.Key)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(f), nil
|
|
||||||
|
|
||||||
case MethodLatestFactBySource:
|
|
||||||
var p keySourceReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
f, err := api.LatestFactBySource(ctx, p.Key, p.Source)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(f), nil
|
|
||||||
|
|
||||||
case MethodSince:
|
|
||||||
var p sinceReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
d, err := api.Since(ctx, p.Key, p.Now)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(sinceResp{Dur: d}), nil
|
|
||||||
|
|
||||||
case MethodPresence:
|
|
||||||
pres, err := api.Presence(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(pres), nil
|
|
||||||
|
|
||||||
case MethodCreateReminder:
|
|
||||||
var p createReminderReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
id, err := api.CreateReminder(ctx, p.Fire, p.Payload, p.Cron)
|
|
||||||
return marshalResult(idResp{ID: id}), err
|
|
||||||
|
|
||||||
case MethodMarkReminder:
|
|
||||||
var p markReminderReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
err := api.MarkReminder(ctx, p.ID, p.Status)
|
|
||||||
return marshalResult(nil), err
|
|
||||||
|
|
||||||
case MethodListReminders:
|
|
||||||
var p nReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.ListReminders(ctx, p.N)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Reminder{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodRecordNudge:
|
|
||||||
var p recordNudgeReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
id, err := api.RecordNudge(ctx, p.Rule, p.Channel, p.Message, p.Ts)
|
|
||||||
return marshalResult(idResp{ID: id}), err
|
|
||||||
|
|
||||||
case MethodResolveNudge:
|
|
||||||
var p resolveNudgeReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
err := api.ResolveNudge(ctx, p.ID, p.Outcome, p.Ts)
|
|
||||||
return marshalResult(nil), err
|
|
||||||
|
|
||||||
case MethodRecentOutcomes:
|
|
||||||
var p outcomesReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.RecentOutcomes(ctx, p.Rule, p.N)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []string{} // stable non-null on the wire
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodRecentFacts:
|
|
||||||
var p nReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.RecentFacts(ctx, p.N)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Fact{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodCalendarEvents:
|
|
||||||
var p calendarEventsReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.CalendarEvents(ctx, p.From, p.To)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Fact{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodRecentNudges:
|
|
||||||
var p nReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.RecentNudges(ctx, p.N)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Nudge{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodWriteNote:
|
|
||||||
var p writeNoteReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
id, err := api.WriteNote(ctx, p.Ts, p.Text, p.Embedding, p.Source)
|
|
||||||
return marshalResult(idResp{ID: id}), err
|
|
||||||
|
|
||||||
case MethodQueryNotes:
|
|
||||||
var p queryNotesReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.QueryNotes(ctx, p.Embedding, p.K)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Note{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodRecentNotes:
|
|
||||||
var p nReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.RecentNotes(ctx, p.N)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Note{}
|
|
||||||
}
|
|
||||||
return marshalResult(out), nil
|
|
||||||
|
|
||||||
case MethodProposeTool:
|
|
||||||
var p proposeToolReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
ok, err := api.ProposeTool(ctx, p.Name, p.Utterance, p.Scope, p.Ts)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(proposeToolResp{Proposed: ok}), nil
|
|
||||||
|
|
||||||
case MethodEnableTool:
|
|
||||||
var p enableToolReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(nil), api.EnableTool(ctx, p.Name, p.Cmd, p.Destructive, p.Scope, p.Ts)
|
|
||||||
|
|
||||||
case MethodDisableTool:
|
|
||||||
var p disableToolReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(nil), api.DisableTool(ctx, p.Name)
|
|
||||||
|
|
||||||
case MethodLookupTool:
|
|
||||||
var p lookupToolReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
t, err := api.LookupTool(ctx, p.Name)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(t), nil
|
|
||||||
|
|
||||||
case MethodListTools:
|
|
||||||
var p listToolsReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
out, err := api.ListTools(ctx, p.Status)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []Tool{}
|
|
||||||
}
|
|
||||||
return marshalResult(listToolsResp{Tools: out}), nil
|
|
||||||
|
|
||||||
case MethodDeleteTool:
|
|
||||||
var p disableToolReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(nil), api.DeleteTool(ctx, p.Name)
|
|
||||||
|
|
||||||
case MethodListProposedRoutines:
|
|
||||||
out, err := api.ListProposedRoutines(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
if out == nil {
|
|
||||||
out = []ProposedRoutine{}
|
|
||||||
}
|
|
||||||
return marshalResult(listProposedRoutinesResp{Routines: out}), nil
|
|
||||||
|
|
||||||
case MethodDismissProposedRoutine:
|
|
||||||
var p dismissProposedRoutineReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(nil), api.DismissProposedRoutine(ctx, p.ID)
|
|
||||||
|
|
||||||
case MethodRevertFact:
|
|
||||||
var p struct {
|
|
||||||
Key string `json:"key"`
|
|
||||||
}
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
newID, err := api.RevertFact(ctx, p.Key)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(map[string]int64{"new_id": newID}), nil
|
|
||||||
|
|
||||||
case MethodChat:
|
|
||||||
var p chatReq
|
|
||||||
if err := unmarshalParams(req.Params, &p); err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
reply, err := api.Chat(ctx, p.Text)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(chatResp{Reply: reply}), nil
|
|
||||||
|
|
||||||
case MethodTickTrace:
|
|
||||||
t, err := api.TickTrace(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(t), nil
|
|
||||||
|
|
||||||
case MethodMorningStatus:
|
|
||||||
s, err := api.MorningStatus(ctx)
|
|
||||||
if err != nil {
|
|
||||||
return nil, err
|
|
||||||
}
|
|
||||||
return marshalResult(s), nil
|
|
||||||
|
|
||||||
case MethodAssertStepUp:
|
case MethodAssertStepUp:
|
||||||
if s.StepUp != nil {
|
if s.StepUp != nil {
|
||||||
return marshalResult(nil), s.StepUp(ctx)
|
return marshalResult(nil), s.StepUp(ctx)
|
||||||
@@ -837,10 +780,13 @@ func (s *Server) dispatch(ctx context.Context, req Request) (json.RawMessage, er
|
|||||||
return marshalResult(nil), s.UnlockFn(ctx, p.PublicKey)
|
return marshalResult(nil), s.UnlockFn(ctx, p.PublicKey)
|
||||||
}
|
}
|
||||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||||
|
}
|
||||||
|
|
||||||
default:
|
h, ok := methodTable[req.Method]
|
||||||
|
if !ok {
|
||||||
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
return nil, fmt.Errorf("%w: %s", ErrUnknownMethod, req.Method)
|
||||||
}
|
}
|
||||||
|
return h(ctx, api, req.Params)
|
||||||
}
|
}
|
||||||
|
|
||||||
func unmarshalParams(raw json.RawMessage, v any) error {
|
func unmarshalParams(raw json.RawMessage, v any) error {
|
||||||
|
|||||||
@@ -0,0 +1,118 @@
|
|||||||
|
package ipc
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ErrNotImplemented is returned by every UnimplementedCoreAPI method. It is
|
||||||
|
// deliberately distinct from ErrUnknownMethod (a wire-level "no such
|
||||||
|
// method exists" verdict) and from any daemon-level "locked" error: this one
|
||||||
|
// means "this method exists on CoreAPI, but the fake/adapter embedding
|
||||||
|
// UnimplementedCoreAPI never got a real implementation for it." A test that
|
||||||
|
// exercises an undeclared method fails loudly on this text instead of
|
||||||
|
// silently nil-panicking or being mistaken for a legitimate failure.
|
||||||
|
var ErrNotImplemented = errors.New("ipc: not implemented (unimplemented CoreAPI stub)")
|
||||||
|
|
||||||
|
// UnimplementedCoreAPI is the gRPC Unimplemented*Server pattern applied to
|
||||||
|
// CoreAPI: embed it in a test double or adapter and override only the
|
||||||
|
// methods you actually exercise. Every method returns ErrNotImplemented, so
|
||||||
|
// a call that reaches an undeclared method fails loudly and specifically,
|
||||||
|
// rather than compiling to a silent no-op or nil-pointer panic. This
|
||||||
|
// replaces the old pattern of hand-writing all 30 no-op stubs per double —
|
||||||
|
// those were compiler-satisfying padding, not tests of anything.
|
||||||
|
type UnimplementedCoreAPI struct{}
|
||||||
|
|
||||||
|
var _ CoreAPI = UnimplementedCoreAPI{}
|
||||||
|
|
||||||
|
func (UnimplementedCoreAPI) WriteFact(ctx context.Context, req WriteFactReq) (int64, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) LatestFact(ctx context.Context, key string) (Fact, error) {
|
||||||
|
return Fact{}, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) LatestFactBySource(ctx context.Context, key, source string) (Fact, error) {
|
||||||
|
return Fact{}, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) Since(ctx context.Context, key string, now time.Time) (time.Duration, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) Presence(ctx context.Context) (Presence, error) {
|
||||||
|
return Presence{}, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) CreateReminder(ctx context.Context, fire time.Time, payload, cron string) (int64, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) MarkReminder(ctx context.Context, id int64, status string) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) ListReminders(ctx context.Context, n int) ([]Reminder, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RecordNudge(ctx context.Context, rule, channel, message string, ts time.Time) (int64, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) ResolveNudge(ctx context.Context, id int64, outcome string, ts time.Time) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RecentOutcomes(ctx context.Context, rule string, n int) ([]string, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RecentFacts(ctx context.Context, n int) ([]Fact, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) CalendarEvents(ctx context.Context, from, to time.Time) ([]Fact, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RecentNudges(ctx context.Context, n int) ([]Nudge, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) QueryNotes(ctx context.Context, embedding []float32, k int) ([]Note, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RecentNotes(ctx context.Context, n int) ([]Note, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) ProposeTool(ctx context.Context, name, utterance, scope string, ts time.Time) (bool, error) {
|
||||||
|
return false, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) EnableTool(ctx context.Context, name string, cmd []string, destructive bool, scope string, ts time.Time) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) DisableTool(ctx context.Context, name string) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) DeleteTool(ctx context.Context, name string) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) ListProposedRoutines(ctx context.Context) ([]ProposedRoutine, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) DismissProposedRoutine(ctx context.Context, id int64) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) AcceptProposedRoutine(ctx context.Context, id int64) error {
|
||||||
|
return ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) LookupTool(ctx context.Context, name string) (Tool, error) {
|
||||||
|
return Tool{}, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) ListTools(ctx context.Context, status string) ([]Tool, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) RevertFact(ctx context.Context, key string) (int64, error) {
|
||||||
|
return 0, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) TickTrace(ctx context.Context) (TickTrace, error) {
|
||||||
|
return TickTrace{}, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) MorningStatus(ctx context.Context) ([]MorningRoutineStatus, error) {
|
||||||
|
return nil, ErrNotImplemented
|
||||||
|
}
|
||||||
|
func (UnimplementedCoreAPI) Chat(ctx context.Context, text string) (string, error) {
|
||||||
|
return "", ErrNotImplemented
|
||||||
|
}
|
||||||
+28
-27
@@ -13,38 +13,39 @@ import (
|
|||||||
type Method string
|
type Method string
|
||||||
|
|
||||||
const (
|
const (
|
||||||
MethodWriteFact Method = "write_fact"
|
MethodWriteFact Method = "write_fact"
|
||||||
MethodLatestFact Method = "latest_fact"
|
MethodLatestFact Method = "latest_fact"
|
||||||
MethodLatestFactBySource Method = "latest_fact_by_source"
|
MethodLatestFactBySource Method = "latest_fact_by_source"
|
||||||
MethodSince Method = "since"
|
MethodSince Method = "since"
|
||||||
MethodPresence Method = "presence"
|
MethodPresence Method = "presence"
|
||||||
MethodCreateReminder Method = "create_reminder"
|
MethodCreateReminder Method = "create_reminder"
|
||||||
MethodMarkReminder Method = "mark_reminder"
|
MethodMarkReminder Method = "mark_reminder"
|
||||||
MethodListReminders Method = "list_reminders"
|
MethodListReminders Method = "list_reminders"
|
||||||
MethodRecordNudge Method = "record_nudge"
|
MethodRecordNudge Method = "record_nudge"
|
||||||
MethodResolveNudge Method = "resolve_nudge"
|
MethodResolveNudge Method = "resolve_nudge"
|
||||||
MethodRecentOutcomes Method = "recent_outcomes"
|
MethodRecentOutcomes Method = "recent_outcomes"
|
||||||
MethodRecentFacts Method = "recent_facts"
|
MethodRecentFacts Method = "recent_facts"
|
||||||
MethodCalendarEvents Method = "calendar_events"
|
MethodCalendarEvents Method = "calendar_events"
|
||||||
MethodRecentNudges Method = "recent_nudges"
|
MethodRecentNudges Method = "recent_nudges"
|
||||||
MethodWriteNote Method = "write_note"
|
MethodWriteNote Method = "write_note"
|
||||||
MethodQueryNotes Method = "query_notes"
|
MethodQueryNotes Method = "query_notes"
|
||||||
MethodRecentNotes Method = "recent_notes"
|
MethodRecentNotes Method = "recent_notes"
|
||||||
MethodProposeTool Method = "propose_tool"
|
MethodProposeTool Method = "propose_tool"
|
||||||
MethodEnableTool Method = "enable_tool"
|
MethodEnableTool Method = "enable_tool"
|
||||||
MethodDisableTool Method = "disable_tool"
|
MethodDisableTool Method = "disable_tool"
|
||||||
MethodAssertStepUp Method = "assert_stepup"
|
MethodAssertStepUp Method = "assert_stepup"
|
||||||
MethodStoreEncryptionKey Method = "store_encryption_key"
|
MethodStoreEncryptionKey Method = "store_encryption_key"
|
||||||
MethodUnlock Method = "unlock"
|
MethodUnlock Method = "unlock"
|
||||||
MethodLookupTool Method = "lookup_tool"
|
MethodLookupTool Method = "lookup_tool"
|
||||||
MethodListTools Method = "list_tools"
|
MethodListTools Method = "list_tools"
|
||||||
MethodDeleteTool Method = "delete_tool"
|
MethodDeleteTool Method = "delete_tool"
|
||||||
MethodListProposedRoutines Method = "list_proposed_routines"
|
MethodListProposedRoutines Method = "list_proposed_routines"
|
||||||
MethodDismissProposedRoutine Method = "dismiss_proposed_routine"
|
MethodDismissProposedRoutine Method = "dismiss_proposed_routine"
|
||||||
|
MethodAcceptProposedRoutine Method = "accept_proposed_routine"
|
||||||
MethodRevertFact Method = "revert_fact"
|
MethodRevertFact Method = "revert_fact"
|
||||||
MethodTickTrace Method = "tick_trace"
|
MethodTickTrace Method = "tick_trace"
|
||||||
MethodMorningStatus Method = "morning_status"
|
MethodMorningStatus Method = "morning_status"
|
||||||
MethodChat Method = "chat"
|
MethodChat Method = "chat"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Request — one frame from module to core. Params is the JSON-encoded argument
|
// Request — one frame from module to core. Params is the JSON-encoded argument
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
// Package kiwix reads a local Kiwix server (offline Wikipedia and friends).
|
||||||
|
//
|
||||||
|
// Why: the resident model is a 0.8B and invents facts. Letting her read a local
|
||||||
|
// article snippet beats letting her recall. Nothing here talks to the internet;
|
||||||
|
// the Kiwix server is on the same box.
|
||||||
|
//
|
||||||
|
// This is search only. Full articles are ~100KB of HTML, far too big for a 4096
|
||||||
|
// token context, so the unit of context is the search snippet (~500 chars).
|
||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/xml"
|
||||||
|
"fmt"
|
||||||
|
"html"
|
||||||
|
"io"
|
||||||
|
"net/http"
|
||||||
|
"net/url"
|
||||||
|
"regexp"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Result is one search hit.
|
||||||
|
type Result struct {
|
||||||
|
Title string // article title, e.g. "Rayleigh scattering"
|
||||||
|
Path string // e.g. /content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering
|
||||||
|
Snippet string // plain text, tags stripped, entities decoded
|
||||||
|
WordCount int // 0 if the server did not say
|
||||||
|
}
|
||||||
|
|
||||||
|
// Client is a Kiwix HTTP client. Boring on purpose: no retries, no cache.
|
||||||
|
type Client struct {
|
||||||
|
base string
|
||||||
|
http *http.Client
|
||||||
|
}
|
||||||
|
|
||||||
|
// New makes a client for a Kiwix base URL like http://127.0.0.1:8034.
|
||||||
|
func New(baseURL string) *Client {
|
||||||
|
return &Client{
|
||||||
|
base: strings.TrimRight(baseURL, "/"),
|
||||||
|
http: &http.Client{Timeout: 10 * time.Second},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search runs a keyword search in one ZIM (book) and returns up to limit hits.
|
||||||
|
//
|
||||||
|
// Ranking is keyword based, not semantic: "Rayleigh scattering" finds the right
|
||||||
|
// article, "why is the sky blue" finds a TV episode. Pass keywords, not questions.
|
||||||
|
func (c *Client) Search(ctx context.Context, pattern, book string, limit int) ([]Result, error) {
|
||||||
|
if limit <= 0 {
|
||||||
|
limit = 5
|
||||||
|
}
|
||||||
|
q := url.Values{}
|
||||||
|
q.Set("pattern", pattern)
|
||||||
|
q.Set("books.name", book)
|
||||||
|
q.Set("format", "xml")
|
||||||
|
q.Set("pageLength", strconv.Itoa(limit))
|
||||||
|
|
||||||
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, c.base+"/search?"+q.Encode(), nil)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
resp, err := c.http.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != http.StatusOK {
|
||||||
|
return nil, fmt.Errorf("kiwix search: http %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
return ParseSearchRSS(resp.Body)
|
||||||
|
}
|
||||||
|
|
||||||
|
// rss mirrors just the bits of the RSS 2.0 reply we use.
|
||||||
|
type rss struct {
|
||||||
|
Items []struct {
|
||||||
|
Title string `xml:"title"`
|
||||||
|
Link string `xml:"link"`
|
||||||
|
// innerxml keeps the <b> match markers so we can strip them ourselves.
|
||||||
|
Description struct {
|
||||||
|
Inner string `xml:",innerxml"`
|
||||||
|
} `xml:"description"`
|
||||||
|
WordCount string `xml:"wordCount"`
|
||||||
|
} `xml:"channel>item"`
|
||||||
|
}
|
||||||
|
|
||||||
|
var tagRE = regexp.MustCompile(`<[^>]*>`)
|
||||||
|
|
||||||
|
// ParseSearchRSS turns a Kiwix search reply into results. Exported so the parser
|
||||||
|
// is testable from a captured response, with no server running.
|
||||||
|
func ParseSearchRSS(r io.Reader) ([]Result, error) {
|
||||||
|
var doc rss
|
||||||
|
if err := xml.NewDecoder(r).Decode(&doc); err != nil {
|
||||||
|
return nil, fmt.Errorf("kiwix search: bad xml: %w", err)
|
||||||
|
}
|
||||||
|
out := make([]Result, 0, len(doc.Items))
|
||||||
|
for _, it := range doc.Items {
|
||||||
|
n, _ := strconv.Atoi(strings.ReplaceAll(it.WordCount, ",", ""))
|
||||||
|
out = append(out, Result{
|
||||||
|
Title: strings.TrimSpace(it.Title),
|
||||||
|
Path: strings.TrimSpace(it.Link),
|
||||||
|
Snippet: plainText(it.Description.Inner),
|
||||||
|
WordCount: n,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// plainText drops markup and decodes entities, leaving text a model can read.
|
||||||
|
func plainText(s string) string {
|
||||||
|
s = tagRE.ReplaceAllString(s, "")
|
||||||
|
s = html.UnescapeString(s)
|
||||||
|
return strings.TrimSpace(strings.Join(strings.Fields(s), " "))
|
||||||
|
}
|
||||||
@@ -0,0 +1,90 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// A real reply from the live server, trimmed to two items.
|
||||||
|
const sampleRSS = `<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<rss version="2.0" xmlns:opensearch="http://a9.com/-/spec/opensearch/1.1/">
|
||||||
|
<channel>
|
||||||
|
<title>Search: Rayleigh scattering</title>
|
||||||
|
<opensearch:totalResults>800</opensearch:totalResults>
|
||||||
|
<item>
|
||||||
|
<title>Rayleigh scattering</title>
|
||||||
|
<link>/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering</link>
|
||||||
|
<description><b>Rayleigh</b> scattering causes the blue color of the sky & yellow colors near the Sun.[1]</description>
|
||||||
|
<book><title>Wikipedia</title></book>
|
||||||
|
<wordCount>2,818</wordCount>
|
||||||
|
</item>
|
||||||
|
<item>
|
||||||
|
<title>Hyper–Rayleigh scattering</title>
|
||||||
|
<link>/content/wikipedia_en_all_maxi_2026-02/Hyper%E2%80%93Rayleigh_scattering</link>
|
||||||
|
<description>...<b>Rayleigh</b> scattering" is a nonlinear optical counterpart.</description>
|
||||||
|
<book><title>Wikipedia</title></book>
|
||||||
|
<wordCount>914</wordCount>
|
||||||
|
</item>
|
||||||
|
</channel>
|
||||||
|
</rss>`
|
||||||
|
|
||||||
|
func TestParseSearchRSS(t *testing.T) {
|
||||||
|
got, err := ParseSearchRSS(strings.NewReader(sampleRSS))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse: %v", err)
|
||||||
|
}
|
||||||
|
if len(got) != 2 {
|
||||||
|
t.Fatalf("want 2 results, got %d", len(got))
|
||||||
|
}
|
||||||
|
if got[0].Title != "Rayleigh scattering" {
|
||||||
|
t.Errorf("title = %q", got[0].Title)
|
||||||
|
}
|
||||||
|
if got[0].Path != "/content/wikipedia_en_all_maxi_2026-02/Rayleigh_scattering" {
|
||||||
|
t.Errorf("path = %q", got[0].Path)
|
||||||
|
}
|
||||||
|
if got[0].WordCount != 2818 {
|
||||||
|
t.Errorf("wordCount = %d", got[0].WordCount)
|
||||||
|
}
|
||||||
|
want := "Rayleigh scattering causes the blue color of the sky & yellow colors near the Sun.[1]"
|
||||||
|
if got[0].Snippet != want {
|
||||||
|
t.Errorf("snippet = %q, want %q", got[0].Snippet, want)
|
||||||
|
}
|
||||||
|
if strings.Contains(got[1].Snippet, "<b>") {
|
||||||
|
t.Errorf("second snippet still has tags: %q", got[1].Snippet)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseSearchRSSBadXML(t *testing.T) {
|
||||||
|
if _, err := ParseSearchRSS(strings.NewReader("not xml at all")); err == nil {
|
||||||
|
t.Fatal("want an error on junk input")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Opt-in: needs a live Kiwix server. CI has none.
|
||||||
|
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 no_proxy=127.0.0.1,localhost go test -run Retrieval -v ./internal/kiwix/
|
||||||
|
func TestRetrievalEval(t *testing.T) {
|
||||||
|
base := os.Getenv("MAVEN_KIWIX_URL")
|
||||||
|
if base == "" {
|
||||||
|
t.Skip("set MAVEN_KIWIX_URL to run the retrieval eval")
|
||||||
|
}
|
||||||
|
noProxyLoopback(t)
|
||||||
|
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
rep, err := RunRetrievalEval(ctx, New(base), 5)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("eval: %v", err)
|
||||||
|
}
|
||||||
|
// No pass bar on purpose: the number is the finding.
|
||||||
|
t.Log("\n" + rep.String() + rep.Detail())
|
||||||
|
}
|
||||||
|
|
||||||
|
// noProxyLoopback stops the box's SOCKS bridge from eating loopback requests.
|
||||||
|
func noProxyLoopback(t *testing.T) {
|
||||||
|
t.Setenv("no_proxy", "127.0.0.1,localhost")
|
||||||
|
t.Setenv("NO_PROXY", "127.0.0.1,localhost")
|
||||||
|
}
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
{
|
||||||
|
"name": "kiwix-knowledge-v1",
|
||||||
|
"book": "wikipedia_en_all_maxi_2026-02",
|
||||||
|
"note": "The 9 knowledge cases from internal/phraser/eval/talk_v1.json. Queries are hand-written English keywords on purpose: Kiwix ranks by keyword, not meaning, so a natural question fails. Writing them by hand separates 'retrieval is broken' from 'the model writes bad queries'.",
|
||||||
|
"cases": [
|
||||||
|
{
|
||||||
|
"id": "know-sky-blue",
|
||||||
|
"question": "почему небо синее?",
|
||||||
|
"query": "Rayleigh scattering sky blue",
|
||||||
|
"want_titles": ["Rayleigh scattering", "Diffuse sky radiation"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-boil-egg",
|
||||||
|
"question": "сколько варить яйцо вкрутую?",
|
||||||
|
"query": "boiled egg cooking",
|
||||||
|
"want_titles": ["Boiled egg", "Egg as food"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-ssd-vs-hdd",
|
||||||
|
"question": "чем ssd отличается от hdd?",
|
||||||
|
"query": "solid-state drive",
|
||||||
|
"want_titles": ["Solid-state drive", "Hard disk drive"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-cat-purr",
|
||||||
|
"question": "почему кошки мурчат?",
|
||||||
|
"query": "cat purr",
|
||||||
|
"want_titles": ["Purr", "Cat communication"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-hiccups",
|
||||||
|
"question": "как быстро избавиться от икоты?",
|
||||||
|
"query": "hiccup",
|
||||||
|
"want_titles": ["Hiccup"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-polite-form",
|
||||||
|
"question": "не могли бы вы объяснить, что такое vpn?",
|
||||||
|
"query": "virtual private network",
|
||||||
|
"want_titles": ["Virtual private network"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-dont-know",
|
||||||
|
"question": "как зовут моего соседа снизу?",
|
||||||
|
"query": "name of my downstairs neighbour",
|
||||||
|
"want_titles": [],
|
||||||
|
"expect_miss": true,
|
||||||
|
"note": "Unanswerable by design. Retrieval SHOULD find nothing useful. Counted as a hit only when nothing relevant comes back."
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-water-per-day",
|
||||||
|
"question": "сколько воды в день надо пить?",
|
||||||
|
"query": "human daily water requirement drinking",
|
||||||
|
"want_titles": ["Drinking water", "Water", "Dehydration", "Hydration"]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "know-thunder-delay",
|
||||||
|
"question": "почему гром слышно позже молнии?",
|
||||||
|
"query": "thunder speed of sound lightning",
|
||||||
|
"want_titles": ["Thunder", "Lightning"]
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
// This scores retrieval alone: no LLM. For each general-knowledge question we
|
||||||
|
// hand-write English keywords and ask whether the article that would answer it
|
||||||
|
// comes back in the top N hits. If this score is low, reading Wikipedia cannot
|
||||||
|
// help the model no matter how good the prompt is.
|
||||||
|
//
|
||||||
|
// The unanswerable case (know-dont-know) is not scored. Whether the junk it
|
||||||
|
// returns is "nothing useful" is a human judgement, so the report just prints
|
||||||
|
// the titles and leaves the score to the 8 answerable cases.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
_ "embed"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
//go:embed knowledge_v1.json
|
||||||
|
var knowledgeFixtureJSON []byte
|
||||||
|
|
||||||
|
// EvalCase — one question with hand-written keywords.
|
||||||
|
type EvalCase struct {
|
||||||
|
ID string `json:"id"`
|
||||||
|
Question string `json:"question"`
|
||||||
|
Query string `json:"query"`
|
||||||
|
WantTitles []string `json:"want_titles"`
|
||||||
|
ExpectMiss bool `json:"expect_miss"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type fixture struct {
|
||||||
|
Name string `json:"name"`
|
||||||
|
Book string `json:"book"`
|
||||||
|
Cases []EvalCase `json:"cases"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// Outcome — what one case retrieved.
|
||||||
|
type Outcome struct {
|
||||||
|
Case EvalCase
|
||||||
|
Titles []string // titles of the top N hits, in rank order
|
||||||
|
Rank int // 1-based rank of the first wanted title, 0 if none
|
||||||
|
Err error
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hit is true when a wanted title came back.
|
||||||
|
func (o Outcome) Hit() bool { return o.Rank > 0 }
|
||||||
|
|
||||||
|
// Report — the score plus per-case detail.
|
||||||
|
type Report struct {
|
||||||
|
Name string
|
||||||
|
Book string
|
||||||
|
TopN int
|
||||||
|
Scored int // answerable cases
|
||||||
|
Hits int
|
||||||
|
Errors int
|
||||||
|
Outcomes []Outcome
|
||||||
|
}
|
||||||
|
|
||||||
|
// Accuracy over the answerable cases.
|
||||||
|
func (r Report) Accuracy() float64 {
|
||||||
|
if r.Scored == 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return float64(r.Hits) / float64(r.Scored)
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunRetrievalEval searches for every fixture case.
|
||||||
|
func RunRetrievalEval(ctx context.Context, c *Client, topN int) (Report, error) {
|
||||||
|
var f fixture
|
||||||
|
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
|
||||||
|
return Report{}, err
|
||||||
|
}
|
||||||
|
rep := Report{Name: f.Name, Book: f.Book, TopN: topN}
|
||||||
|
for _, cs := range f.Cases {
|
||||||
|
res, err := c.Search(ctx, cs.Query, f.Book, topN)
|
||||||
|
o := Outcome{Case: cs, Err: err}
|
||||||
|
if err != nil {
|
||||||
|
rep.Errors++
|
||||||
|
}
|
||||||
|
for i, hit := range res {
|
||||||
|
o.Titles = append(o.Titles, hit.Title)
|
||||||
|
if o.Rank == 0 && matches(cs.WantTitles, hit.Title) {
|
||||||
|
o.Rank = i + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !cs.ExpectMiss {
|
||||||
|
rep.Scored++
|
||||||
|
if o.Hit() {
|
||||||
|
rep.Hits++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
rep.Outcomes = append(rep.Outcomes, o)
|
||||||
|
}
|
||||||
|
return rep, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func matches(want []string, title string) bool {
|
||||||
|
for _, w := range want {
|
||||||
|
if strings.EqualFold(strings.TrimSpace(title), w) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// String — the headline number.
|
||||||
|
func (r Report) String() string {
|
||||||
|
var b strings.Builder
|
||||||
|
fmt.Fprintf(&b, "%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n",
|
||||||
|
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors)
|
||||||
|
fmt.Fprintf(&b, " book: %s\n", r.Book)
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detail — per case: what was asked, what was searched, what came back.
|
||||||
|
func (r Report) Detail() string {
|
||||||
|
var b strings.Builder
|
||||||
|
for _, o := range r.Outcomes {
|
||||||
|
mark := "MISS"
|
||||||
|
switch {
|
||||||
|
case o.Case.ExpectMiss:
|
||||||
|
mark = "n/a "
|
||||||
|
case o.Hit():
|
||||||
|
mark = fmt.Sprintf("hit@%d", o.Rank)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " %-6s %-20s q=%q\n", mark, o.Case.ID, o.Case.Query)
|
||||||
|
if o.Err != nil {
|
||||||
|
fmt.Fprintf(&b, " error: %v\n", o.Err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
@@ -0,0 +1,183 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
// Turning a Russian question into an English Kiwix search.
|
||||||
|
//
|
||||||
|
// Kiwix ranks by keyword, not by meaning. "why is the sky blue" returns a TV
|
||||||
|
// episode; "Rayleigh scattering sky blue" returns the right article. So the
|
||||||
|
// model's job here is NOT translation — it is naming the English article the
|
||||||
|
// answer lives in.
|
||||||
|
//
|
||||||
|
// The output space is a handful of words, so it is worth locking down hard: a
|
||||||
|
// GBNF grammar for the shape, a tiny token cap, and a cleanup pass that throws
|
||||||
|
// away anything odd rather than handing junk to Kiwix.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
"unicode"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Completer — the LLM seam, so tests can fake it. *llm.Client satisfies it.
|
||||||
|
type Completer interface {
|
||||||
|
Complete(ctx context.Context, r llm.Req) (string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// queryGrammar — one JSON object holding 1..6 keyword words. Latin letters,
|
||||||
|
// digits and hyphens only, so the model physically cannot answer the question
|
||||||
|
// or reply in Russian.
|
||||||
|
//
|
||||||
|
// Why the JSON wrapper: this model always thinks out loud and this llama-server
|
||||||
|
// build ignores the thinking switch (see ROUTING-EVAL-31-07-2026.md). A bare
|
||||||
|
// word-list grammar just captured the reasoning — every case came back as
|
||||||
|
// "Let me analyze this request carefully". Demanding JSON, like routeGrammar and
|
||||||
|
// responseGrammar already do, gives the reasoning nowhere to go.
|
||||||
|
const queryGrammar = `
|
||||||
|
root ::= "{" ws "\"query\"" ws ":" ws "\"" word (" " word){0,5} "\"" ws "}"
|
||||||
|
word ::= [A-Za-z0-9] [A-Za-z0-9-]{0,23}
|
||||||
|
ws ::= [ \t\n]*
|
||||||
|
`
|
||||||
|
|
||||||
|
// rewriteSystem — asks for search keywords, not an answer and not a translation.
|
||||||
|
const rewriteSystem = `You turn a question into a search query for English Wikipedia.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Output ONLY English search keywords. Never an answer, never an explanation.
|
||||||
|
- Do NOT translate the sentence. Name the thing the answer is about.
|
||||||
|
- The output must be a noun phrase, like a Wikipedia article title.
|
||||||
|
- Never use question words: no why, how, what, when, which, "how much",
|
||||||
|
"how long", "how to", "vs", "reason", "difference".
|
||||||
|
- 2 to 4 words.
|
||||||
|
|
||||||
|
Reply with JSON: {"query":"<keywords>"}
|
||||||
|
|
||||||
|
Good:
|
||||||
|
"почему листья желтеют осенью?" -> {"query":"leaf senescence autumn"}
|
||||||
|
"как работает микроволновка?" -> {"query":"microwave oven"}
|
||||||
|
"не могли бы вы объяснить, что такое блокчейн?" -> {"query":"blockchain"}
|
||||||
|
"сколько живут собаки?" -> {"query":"dog lifespan"}
|
||||||
|
"как избавиться от комаров в квартире?" -> {"query":"mosquito control"}
|
||||||
|
"чем чай отличается от кофе?" -> {"query":"tea"}
|
||||||
|
|
||||||
|
Only JSON, no explanation.`
|
||||||
|
|
||||||
|
// maxQueryTokens — the output is a few words plus the JSON wrapper. A tight cap
|
||||||
|
// is the cheapest guard against the model rambling into an answer.
|
||||||
|
const maxQueryTokens = 32
|
||||||
|
|
||||||
|
// Rewriter asks the resident model for English search keywords.
|
||||||
|
type Rewriter struct{ c Completer }
|
||||||
|
|
||||||
|
func NewRewriter(c Completer) *Rewriter { return &Rewriter{c: c} }
|
||||||
|
|
||||||
|
// Rewrite returns English keywords for a question in any language.
|
||||||
|
// It errors rather than returning something Kiwix should not see.
|
||||||
|
func (r *Rewriter) Rewrite(ctx context.Context, question string) (string, error) {
|
||||||
|
raw, err := r.c.Complete(ctx, llm.Req{
|
||||||
|
System: rewriteSystem,
|
||||||
|
User: strings.TrimSpace(question),
|
||||||
|
Grammar: queryGrammar,
|
||||||
|
MaxTokens: maxQueryTokens,
|
||||||
|
RepeatPenalty: 1.15,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
return CleanQuery(unwrapJSON(raw))
|
||||||
|
}
|
||||||
|
|
||||||
|
// unwrapJSON pulls the query out of {"query":"..."}. If the reply is not that
|
||||||
|
// shape it is returned as-is, and CleanQuery decides whether it is usable.
|
||||||
|
func unwrapJSON(raw string) string {
|
||||||
|
s := strings.TrimSpace(raw)
|
||||||
|
if !strings.HasPrefix(s, "{") {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
var got struct{ Query string }
|
||||||
|
if err := json.Unmarshal([]byte(s), &got); err != nil {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return got.Query
|
||||||
|
}
|
||||||
|
|
||||||
|
// maxQueryWords matches the grammar's bound. Anything longer is prose.
|
||||||
|
const maxQueryWords = 6
|
||||||
|
|
||||||
|
// CleanQuery checks and tidies whatever the model produced. The grammar makes
|
||||||
|
// bad output unlikely, not impossible (a server without grammar support, a
|
||||||
|
// different model), so this is the real gate in front of Kiwix.
|
||||||
|
//
|
||||||
|
// Exported so it can be tested without a model.
|
||||||
|
func CleanQuery(raw string) (string, error) {
|
||||||
|
s := strings.TrimSpace(raw)
|
||||||
|
// Models like to wrap answers in quotes. Drop surrounding ones.
|
||||||
|
s = strings.Trim(s, "\"'`")
|
||||||
|
// Keep the first line only: everything after it is prose.
|
||||||
|
if i := strings.IndexAny(s, "\r\n"); i >= 0 {
|
||||||
|
s = s[:i]
|
||||||
|
}
|
||||||
|
// Keep letters, digits, spaces and hyphens; anything else becomes a space.
|
||||||
|
var b strings.Builder
|
||||||
|
for _, ru := range s {
|
||||||
|
switch {
|
||||||
|
case unicode.IsLetter(ru) || unicode.IsDigit(ru) || ru == '-':
|
||||||
|
b.WriteRune(ru)
|
||||||
|
default:
|
||||||
|
b.WriteRune(' ')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
words := strings.Fields(b.String())
|
||||||
|
if len(words) == 0 {
|
||||||
|
return "", fmt.Errorf("kiwix rewrite: empty query")
|
||||||
|
}
|
||||||
|
if len(words) > maxQueryWords {
|
||||||
|
return "", fmt.Errorf("kiwix rewrite: %d words, want at most %d (looks like prose)", len(words), maxQueryWords)
|
||||||
|
}
|
||||||
|
words = dropStopWords(words)
|
||||||
|
out := strings.Join(words, " ")
|
||||||
|
// The ZIMs are English. Non-Latin letters mean the model ignored the ask.
|
||||||
|
for _, ru := range out {
|
||||||
|
if unicode.IsLetter(ru) && !isLatin(ru) {
|
||||||
|
return "", fmt.Errorf("kiwix rewrite: query is not English: %q", out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// stopWords — question words and filler. The model keeps writing question-shaped
|
||||||
|
// queries ("why is the sky blue", "how much water to drink daily") no matter how
|
||||||
|
// the prompt is worded, and Kiwix ranks on every word, so those words drag in
|
||||||
|
// song and episode titles. Dropping them in code is not a style preference: a
|
||||||
|
// keyword ranker gets nothing from them.
|
||||||
|
var stopWords = map[string]bool{
|
||||||
|
"a": true, "an": true, "the": true, "is": true, "are": true, "was": true,
|
||||||
|
"do": true, "does": true, "did": true, "to": true, "of": true, "in": true,
|
||||||
|
"on": true, "for": true, "and": true, "or": true, "my": true, "me": true,
|
||||||
|
"i": true, "it": true, "its": true, "be": true, "been": true, "get": true,
|
||||||
|
"how": true, "why": true, "what": true, "when": true, "which": true,
|
||||||
|
"who": true, "where": true, "much": true, "many": true, "long": true,
|
||||||
|
"vs": true, "than": true, "rid": true, "from": true, "about": true,
|
||||||
|
}
|
||||||
|
|
||||||
|
// dropStopWords removes filler, but never everything: if the query was nothing
|
||||||
|
// but stop words there is nothing better to search, so the original is kept and
|
||||||
|
// the caller sees whatever Kiwix makes of it.
|
||||||
|
func dropStopWords(words []string) []string {
|
||||||
|
kept := make([]string, 0, len(words))
|
||||||
|
for _, w := range words {
|
||||||
|
if !stopWords[strings.ToLower(w)] {
|
||||||
|
kept = append(kept, w)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(kept) == 0 {
|
||||||
|
return words
|
||||||
|
}
|
||||||
|
return kept
|
||||||
|
}
|
||||||
|
|
||||||
|
func isLatin(ru rune) bool {
|
||||||
|
return (ru >= 'a' && ru <= 'z') || (ru >= 'A' && ru <= 'Z')
|
||||||
|
}
|
||||||
@@ -0,0 +1,96 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
// End-to-end score: Russian question -> model rewrite -> Kiwix search -> did a
|
||||||
|
// wanted article come back. Same 9 cases as the retrieval eval, so the two
|
||||||
|
// numbers are directly comparable: retrieval with hand-written keywords is the
|
||||||
|
// ceiling, this is what the model actually reaches.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
// RewriteOutcome — one case, end to end.
|
||||||
|
type RewriteOutcome struct {
|
||||||
|
Outcome
|
||||||
|
ModelQuery string // what the model asked for ("" if it failed)
|
||||||
|
RewriteErr error
|
||||||
|
}
|
||||||
|
|
||||||
|
// RunRewriteEval rewrites every question with the model, then searches.
|
||||||
|
func RunRewriteEval(ctx context.Context, c *Client, rw *Rewriter, topN int) (RewriteReport, error) {
|
||||||
|
var f fixture
|
||||||
|
if err := json.Unmarshal(knowledgeFixtureJSON, &f); err != nil {
|
||||||
|
return RewriteReport{}, err
|
||||||
|
}
|
||||||
|
rep := RewriteReport{Report: Report{Name: f.Name + "-rewrite", Book: f.Book, TopN: topN}}
|
||||||
|
for _, cs := range f.Cases {
|
||||||
|
out := RewriteOutcome{Outcome: Outcome{Case: cs}}
|
||||||
|
q, err := rw.Rewrite(ctx, cs.Question)
|
||||||
|
out.ModelQuery, out.RewriteErr = q, err
|
||||||
|
if err == nil {
|
||||||
|
res, serr := c.Search(ctx, q, f.Book, topN)
|
||||||
|
out.Err = serr
|
||||||
|
for i, hit := range res {
|
||||||
|
out.Titles = append(out.Titles, hit.Title)
|
||||||
|
if out.Rank == 0 && matches(cs.WantTitles, hit.Title) {
|
||||||
|
out.Rank = i + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if out.RewriteErr != nil || out.Err != nil {
|
||||||
|
rep.Errors++
|
||||||
|
}
|
||||||
|
if !cs.ExpectMiss {
|
||||||
|
rep.Scored++
|
||||||
|
if out.Hit() {
|
||||||
|
rep.Hits++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
rep.Cases = append(rep.Cases, out)
|
||||||
|
}
|
||||||
|
return rep, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// RewriteReport — the score plus per-case detail.
|
||||||
|
type RewriteReport struct {
|
||||||
|
Report
|
||||||
|
Cases []RewriteOutcome
|
||||||
|
}
|
||||||
|
|
||||||
|
// String — the headline number.
|
||||||
|
func (r RewriteReport) String() string {
|
||||||
|
return fmt.Sprintf("%s: %d/%d answerable questions retrieve a wanted article in top %d (%.1f%%), %d errors\n book: %s\n",
|
||||||
|
r.Name, r.Hits, r.Scored, r.TopN, 100*r.Accuracy(), r.Errors, r.Book)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detail — per case: hand-written query next to the model's, and what came back.
|
||||||
|
// The point is seeing WHERE the model's phrasing differs, not just the score.
|
||||||
|
func (r RewriteReport) Detail() string {
|
||||||
|
var b strings.Builder
|
||||||
|
for _, o := range r.Cases {
|
||||||
|
mark := "MISS"
|
||||||
|
switch {
|
||||||
|
case o.Case.ExpectMiss:
|
||||||
|
mark = "n/a "
|
||||||
|
case o.Hit():
|
||||||
|
mark = fmt.Sprintf("hit@%d", o.Rank)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " %-6s %-20s\n", mark, o.Case.ID)
|
||||||
|
fmt.Fprintf(&b, " asked: %s\n", o.Case.Question)
|
||||||
|
fmt.Fprintf(&b, " hand: %q\n", o.Case.Query)
|
||||||
|
fmt.Fprintf(&b, " model: %q\n", o.ModelQuery)
|
||||||
|
if o.RewriteErr != nil {
|
||||||
|
fmt.Fprintf(&b, " rewrite rejected: %v\n", o.RewriteErr)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if o.Err != nil {
|
||||||
|
fmt.Fprintf(&b, " search error: %v\n", o.Err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " got: %s\n", strings.Join(o.Titles, " | "))
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Opt-in: needs a live Kiwix server AND a live llama-server.
|
||||||
|
// MAVEN_KIWIX_URL=http://127.0.0.1:8034 MAVEN_LLM_URL=http://127.0.0.1:18099 \
|
||||||
|
//
|
||||||
|
// no_proxy=127.0.0.1,localhost go test -run RewriteEval -v ./internal/kiwix/
|
||||||
|
func TestRewriteEval(t *testing.T) {
|
||||||
|
kbase, lbase := os.Getenv("MAVEN_KIWIX_URL"), os.Getenv("MAVEN_LLM_URL")
|
||||||
|
if kbase == "" || lbase == "" {
|
||||||
|
t.Skip("set MAVEN_KIWIX_URL and MAVEN_LLM_URL to run the rewrite eval")
|
||||||
|
}
|
||||||
|
noProxyLoopback(t)
|
||||||
|
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
rw := NewRewriter(llm.New(lbase, 3*time.Minute))
|
||||||
|
rep, err := RunRewriteEval(ctx, New(kbase), rw, 5)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("eval: %v", err)
|
||||||
|
}
|
||||||
|
// No pass bar on purpose: the number is the finding.
|
||||||
|
t.Log("\n" + rep.String() + rep.Detail())
|
||||||
|
}
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
package kiwix
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Bad model output must never reach Kiwix. No model needed for this.
|
||||||
|
func TestCleanQueryRejectsJunk(t *testing.T) {
|
||||||
|
bad := []struct{ name, raw string }{
|
||||||
|
{"empty", ""},
|
||||||
|
{"blank", " \n "},
|
||||||
|
{"russian came back", "почему небо синее"},
|
||||||
|
{"mixed russian", "sky синее scattering"},
|
||||||
|
{"full sentence", "The sky looks blue because of the scattering of sunlight by air molecules"},
|
||||||
|
{"prose with quotes", `Sure! Here is a good search query: "Rayleigh scattering", which explains it.`},
|
||||||
|
}
|
||||||
|
for _, c := range bad {
|
||||||
|
if got, err := CleanQuery(c.raw); err == nil {
|
||||||
|
t.Errorf("%s: want rejection, got %q", c.name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestCleanQueryCleans(t *testing.T) {
|
||||||
|
ok := []struct{ raw, want string }{
|
||||||
|
{"Rayleigh scattering sky", "Rayleigh scattering sky"},
|
||||||
|
{" boiled egg cooking \n", "boiled egg cooking"},
|
||||||
|
{`"virtual private network"`, "virtual private network"},
|
||||||
|
{"solid-state drive", "solid-state drive"},
|
||||||
|
{"cat purr.", "cat purr"},
|
||||||
|
{"hiccup\nAlso: hiccough", "hiccup"},
|
||||||
|
// Question words are filler to a keyword ranker, so they go.
|
||||||
|
{"why is the sky blue", "sky blue"},
|
||||||
|
{"how much water to drink daily", "water drink daily"},
|
||||||
|
{"SSD vs HDD comparison", "SSD HDD comparison"},
|
||||||
|
// Nothing but filler: keep it rather than return nothing.
|
||||||
|
{"what is it", "what is it"},
|
||||||
|
}
|
||||||
|
for _, c := range ok {
|
||||||
|
got, err := CleanQuery(c.raw)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("%q: %v", c.raw, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if got != c.want {
|
||||||
|
t.Errorf("%q -> %q, want %q", c.raw, got, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
type fakeCompleter struct {
|
||||||
|
out string
|
||||||
|
req llm.Req
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeCompleter) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||||
|
f.req = r
|
||||||
|
return f.out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRewriteConstrainsTheCall(t *testing.T) {
|
||||||
|
f := &fakeCompleter{out: `{"query":"Rayleigh scattering sky"}`}
|
||||||
|
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("rewrite: %v", err)
|
||||||
|
}
|
||||||
|
if got != "Rayleigh scattering sky" {
|
||||||
|
t.Errorf("query = %q", got)
|
||||||
|
}
|
||||||
|
if f.req.Grammar == "" {
|
||||||
|
t.Error("no grammar sent")
|
||||||
|
}
|
||||||
|
if f.req.MaxTokens == 0 || f.req.MaxTokens > 32 {
|
||||||
|
t.Errorf("max_tokens = %d, want a small cap", f.req.MaxTokens)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRewriteRejectsBadModelOutput(t *testing.T) {
|
||||||
|
bad := []string{
|
||||||
|
`{"query":"почему небо синее"}`, // never translated
|
||||||
|
`{"query":""}`, // empty
|
||||||
|
`{"query":"the sky is blue because sunlight is scattered by air"}`, // an answer
|
||||||
|
// Note: a SHORT English prose fragment ("Let me analyze this request")
|
||||||
|
// is under the word cap and cannot be caught here. The grammar is what
|
||||||
|
// stops that one.
|
||||||
|
}
|
||||||
|
for _, out := range bad {
|
||||||
|
f := &fakeCompleter{out: out}
|
||||||
|
if got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?"); err == nil {
|
||||||
|
t.Errorf("%s: want rejection, got %q", out, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A reply that is not the JSON shape but is still usable keywords should pass.
|
||||||
|
func TestRewriteFallsBackToPlainText(t *testing.T) {
|
||||||
|
f := &fakeCompleter{out: "Rayleigh scattering sky"}
|
||||||
|
got, err := NewRewriter(f).Rewrite(context.Background(), "почему небо синее?")
|
||||||
|
if err != nil || got != "Rayleigh scattering sky" {
|
||||||
|
t.Errorf("got %q, %v", got, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -19,7 +19,7 @@ func TestComplete(t *testing.T) {
|
|||||||
t.Errorf("path = %q, want /v1/chat/completions", r.URL.Path)
|
t.Errorf("path = %q, want /v1/chat/completions", r.URL.Path)
|
||||||
}
|
}
|
||||||
var reqBody struct {
|
var reqBody struct {
|
||||||
Messages []struct {
|
Messages []struct {
|
||||||
Role string `json:"role"`
|
Role string `json:"role"`
|
||||||
Content string `json:"content"`
|
Content string `json:"content"`
|
||||||
} `json:"messages"`
|
} `json:"messages"`
|
||||||
|
|||||||
@@ -0,0 +1,70 @@
|
|||||||
|
package llm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"net/http"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// UnknownModel is the label to print when the server would not say what it has
|
||||||
|
// loaded. Deliberately ugly: an honest "unknown" is fine, a plausible-looking
|
||||||
|
// but wrong model name is the bug this whole file exists to prevent.
|
||||||
|
const UnknownModel = "unknown-model"
|
||||||
|
|
||||||
|
// llama-server is local, so never send this through a proxy: this box's
|
||||||
|
// http_proxy answers 503 for loopback, which would look like "server won't say
|
||||||
|
// which model it has" when the server is right there and fine.
|
||||||
|
// A Transport with no Proxy set bypasses http_proxy entirely.
|
||||||
|
var modelHTTP = &http.Client{Timeout: 10 * time.Second, Transport: &http.Transport{}}
|
||||||
|
|
||||||
|
// ModelID asks llama-server which model it has loaded, so a scoring run can
|
||||||
|
// label itself. Without this a bake-off between two models produces two tables
|
||||||
|
// that look identical, and the operator has to remember which server was up.
|
||||||
|
//
|
||||||
|
// Read from the server rather than passed in on purpose: a hand-typed label
|
||||||
|
// goes stale the moment someone restarts the server with a different -m.
|
||||||
|
func ModelID(ctx context.Context, base string) (string, error) {
|
||||||
|
req, err := http.NewRequestWithContext(ctx, "GET", strings.TrimSuffix(base, "/")+"/v1/models", nil)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
resp, err := modelHTTP.Do(req)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
defer resp.Body.Close()
|
||||||
|
if resp.StatusCode != 200 {
|
||||||
|
return "", fmt.Errorf("models: status %d", resp.StatusCode)
|
||||||
|
}
|
||||||
|
var out struct {
|
||||||
|
Data []struct {
|
||||||
|
ID string `json:"id"`
|
||||||
|
} `json:"data"`
|
||||||
|
}
|
||||||
|
if err := json.NewDecoder(resp.Body).Decode(&out); err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
if len(out.Data) == 0 {
|
||||||
|
return "", fmt.Errorf("models: empty list")
|
||||||
|
}
|
||||||
|
short := shortModelID(out.Data[0].ID)
|
||||||
|
if short == "" {
|
||||||
|
// Server answered but the id field was missing or blank. Say so
|
||||||
|
// instead of handing back an empty label that reads as a real name.
|
||||||
|
return "", fmt.Errorf("models: no id in response")
|
||||||
|
}
|
||||||
|
return short, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// shortModelID trims the path and the .gguf suffix — llama-server reports the
|
||||||
|
// file name it was started with, which is too long for a table header.
|
||||||
|
func shortModelID(id string) string {
|
||||||
|
id = strings.TrimSpace(id)
|
||||||
|
if i := strings.LastIndexAny(id, "/\\"); i >= 0 {
|
||||||
|
id = id[i+1:]
|
||||||
|
}
|
||||||
|
return strings.TrimSpace(strings.TrimSuffix(id, ".gguf"))
|
||||||
|
}
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
package llm
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"net/http"
|
||||||
|
"net/http/httptest"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The point of these tests: a wrong-but-plausible model label is the bug, so
|
||||||
|
// every path that cannot learn the real name must return an error instead of a
|
||||||
|
// guess. No llama-server needed — a stub server stands in.
|
||||||
|
func TestModelID(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
body string
|
||||||
|
code int
|
||||||
|
want string // "" ⇒ expect an error
|
||||||
|
}{
|
||||||
|
{"full path", `{"data":[{"id":"/mnt/hdd1/llms/qwen3.5/Qwen3.5-0.8B.Q4_K_M.gguf"}]}`, 200, "Qwen3.5-0.8B.Q4_K_M"},
|
||||||
|
{"bare name", `{"data":[{"id":"LFM2.5-1.2B"}]}`, 200, "LFM2.5-1.2B"},
|
||||||
|
{"empty list", `{"data":[]}`, 200, ""},
|
||||||
|
{"id missing", `{"data":[{}]}`, 200, ""},
|
||||||
|
{"id blank", `{"data":[{"id":" "}]}`, 200, ""},
|
||||||
|
{"server error", `nope`, 500, ""},
|
||||||
|
{"not json", `<html>`, 200, ""},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||||
|
if r.URL.Path != "/v1/models" {
|
||||||
|
t.Errorf("asked for %s, want /v1/models", r.URL.Path)
|
||||||
|
}
|
||||||
|
w.WriteHeader(c.code)
|
||||||
|
_, _ = w.Write([]byte(c.body))
|
||||||
|
}))
|
||||||
|
defer srv.Close()
|
||||||
|
|
||||||
|
got, err := ModelID(context.Background(), srv.URL+"/")
|
||||||
|
if c.want == "" {
|
||||||
|
if err == nil {
|
||||||
|
t.Fatalf("want an error, got label %q", got)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ModelID: %v", err)
|
||||||
|
}
|
||||||
|
if got != c.want {
|
||||||
|
t.Errorf("got %q, want %q", got, c.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestModelIDUnreachable(t *testing.T) {
|
||||||
|
srv := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {}))
|
||||||
|
url := srv.URL
|
||||||
|
srv.Close() // nothing listening now
|
||||||
|
|
||||||
|
if got, err := ModelID(context.Background(), url); err == nil {
|
||||||
|
t.Fatalf("want an error from a dead server, got label %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,358 @@
|
|||||||
|
package loop
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Tests for the universal restraint gate.
|
||||||
|
//
|
||||||
|
// DESIGN.md § Trigger model: "the gate is universal, applied by the loop, never
|
||||||
|
// per-rule — quiet-hours, presence, cooldown, snooze, calendar-busy all live in
|
||||||
|
// one fires()." These tests pin the CONSERVATIVE side of that: the cases where
|
||||||
|
// Maven must stay quiet. They exist so nobody loosens the gate by accident.
|
||||||
|
//
|
||||||
|
// Where the code does not yet do what DESIGN.md promises, the test is written to
|
||||||
|
// show the gap and then skipped, with the file and line to fix. Behaviour is not
|
||||||
|
// changed to make a test pass.
|
||||||
|
|
||||||
|
// testRule — a rule at the given severity that always wants to fire, so the
|
||||||
|
// only thing under test is the gate.
|
||||||
|
func testRule(name string, sev Severity) Rule {
|
||||||
|
return Rule{
|
||||||
|
Name: name,
|
||||||
|
Severity: sev,
|
||||||
|
Cooldown: Cooldown{Base: 30 * time.Minute, Min: time.Minute, Max: time.Hour},
|
||||||
|
Predicate: func(State) bool { return true },
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- quiet hours ------------------------------------
|
||||||
|
|
||||||
|
// Quiet hours silence care and leave ops alone. A failed backup at 2am matters;
|
||||||
|
// a water nudge at 2am does not.
|
||||||
|
func TestGateQuietHoursSuppressesCareOnly(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
sev Severity
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{Sev1, false},
|
||||||
|
{Sev2, false},
|
||||||
|
{Sev3, true},
|
||||||
|
{Sev4, true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
s := State{Now: refTime(), Presence: store.Present, QuietHours: true}
|
||||||
|
if got := Gate(s, testRule("r", c.sev)); got != c.want {
|
||||||
|
t.Errorf("quiet hours sev%d: want fire=%v, got %v", c.sev, c.want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- presence ---------------------------------------
|
||||||
|
|
||||||
|
// DESIGN.md § Delivery: "sev <= 2 drops on away, sev >= 3 holds: a missed water
|
||||||
|
// nudge is noise, a missed backup failure isn't."
|
||||||
|
func TestGateAwayDropsCareHoldsOps(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
sev Severity
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{Sev1, false},
|
||||||
|
{Sev2, false},
|
||||||
|
{Sev3, true},
|
||||||
|
{Sev4, true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
s := State{Now: refTime(), Presence: store.Away}
|
||||||
|
if got := Gate(s, testRule("r", c.sev)); got != c.want {
|
||||||
|
t.Errorf("away sev%d: want fire=%v, got %v", c.sev, c.want, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Care nudges are allowed through when the user is actually there and nothing
|
||||||
|
// else is suppressing. Without this the "quiet" tests above could pass on a
|
||||||
|
// gate that simply never fires.
|
||||||
|
func TestGateAllowsCareWhenPresentAndClear(t *testing.T) {
|
||||||
|
s := State{Now: refTime(), Presence: store.Present}
|
||||||
|
if !Gate(s, testRule("r", Sev1)) {
|
||||||
|
t.Fatal("present and clear: care nudge should be allowed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- calendar busy ----------------------------------
|
||||||
|
|
||||||
|
// "Don't nag mid-meeting" is an env predicate in the gate, not the LLM's call.
|
||||||
|
// Ops still gets through — a service being down mid-meeting is worth the
|
||||||
|
// interruption.
|
||||||
|
func TestGateCalendarBusySuppressesCareOnly(t *testing.T) {
|
||||||
|
care := State{Now: refTime(), Presence: store.Present, CalendarBusy: true}
|
||||||
|
if Gate(care, testRule("r", Sev2)) {
|
||||||
|
t.Error("calendar busy: care nudge should be suppressed")
|
||||||
|
}
|
||||||
|
if !Gate(care, testRule("r", Sev4)) {
|
||||||
|
t.Error("calendar busy: ops hard should still fire")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- cooldown ---------------------------------------
|
||||||
|
|
||||||
|
// Cooldown holds for every severity — it is the anti-nag knob, so ops cannot
|
||||||
|
// buy its way past it either.
|
||||||
|
func TestGateCooldownHoldsForAllSeverities(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
for _, sev := range []Severity{Sev1, Sev2, Sev3, Sev4} {
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
CooldownUntil: map[string]time.Time{"r": now.Add(10 * time.Minute)},
|
||||||
|
}
|
||||||
|
if Gate(s, testRule("r", sev)) {
|
||||||
|
t.Errorf("cooldown sev%d: should be suppressed", sev)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cooldown is per-rule: one rule cooling down must not mute another.
|
||||||
|
func TestGateCooldownIsPerRule(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
CooldownUntil: map[string]time.Time{"water": now.Add(10 * time.Minute)},
|
||||||
|
}
|
||||||
|
if Gate(s, testRule("water", Sev1)) {
|
||||||
|
t.Error("water is cooling down and should be suppressed")
|
||||||
|
}
|
||||||
|
if !Gate(s, testRule("meal", Sev1)) {
|
||||||
|
t.Error("meal has no cooldown and should be allowed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The moment the cooldown expires the rule is free again — the gate compares
|
||||||
|
// with Before, so "until" itself is already clear.
|
||||||
|
func TestGateCooldownExpires(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
CooldownUntil: map[string]time.Time{"r": now},
|
||||||
|
}
|
||||||
|
if !Gate(s, testRule("r", Sev1)) {
|
||||||
|
t.Fatal("cooldown at exactly now should already be clear")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- snooze -----------------------------------------
|
||||||
|
|
||||||
|
// Snooze is the user saying "not about this". It beats everything, including
|
||||||
|
// ops hard.
|
||||||
|
func TestGateSnoozeHoldsForAllSeverities(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
for _, sev := range []Severity{Sev1, Sev2, Sev3, Sev4} {
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
SnoozeUntil: map[string]time.Time{"r": now.Add(time.Hour)},
|
||||||
|
}
|
||||||
|
if Gate(s, testRule("r", sev)) {
|
||||||
|
t.Errorf("snooze sev%d: should be suppressed", sev)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- no-data backstop -------------------------------
|
||||||
|
|
||||||
|
// The gate enforces no-data inertness a second time, for any rule that declared
|
||||||
|
// the keys it needs. A predicate that forgets the check still cannot fire.
|
||||||
|
func TestGateNoDataBackstopBeatsAnEagerPredicate(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
eager := Rule{
|
||||||
|
Name: "eager",
|
||||||
|
Severity: Sev4, // even ops hard does not get past missing data
|
||||||
|
Predicate: func(State) bool { return true },
|
||||||
|
InertWhenNoData: []string{"water", "meal"},
|
||||||
|
}
|
||||||
|
// one of the two keys present is not enough.
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
Facts: map[string]store.Fact{"water": ago("water", "tap:water", `"250ml"`, time.Hour)},
|
||||||
|
}
|
||||||
|
if Gate(s, eager) {
|
||||||
|
t.Fatal("a rule missing one of its keys must stay inert")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- one nudge per tick -----------------------------
|
||||||
|
|
||||||
|
// All five default rules want to fire at once. The tick must still emit exactly
|
||||||
|
// one candidate, the loudest — never a dogpile.
|
||||||
|
func TestTickNeverDogpilesAndPicksLoudest(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
Facts: map[string]store.Fact{
|
||||||
|
"water": ago("water", "tap:water", `"250ml"`, 5*time.Hour),
|
||||||
|
"meal": ago("meal", "voice", `"lunch"`, 8*time.Hour),
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 30*time.Second),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 3*time.Hour),
|
||||||
|
"service_down": ago("service_down", "poll:uptimekuma", `"down"`, time.Minute),
|
||||||
|
"netdata_alarm": ago("netdata_alarm", "poll:netdata", `"critical"`, time.Minute),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
// sanity: every rule really does want to fire, so the pick is a real choice.
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
if !r.Predicate(s) {
|
||||||
|
t.Fatalf("setup: rule %q does not want to fire", r.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
got := Tick(s, DefaultRules())
|
||||||
|
if got == nil {
|
||||||
|
t.Fatal("all rules firing: want one candidate, got nil")
|
||||||
|
}
|
||||||
|
if got.Rule.Name != "service_down" || got.Severity != Sev4 {
|
||||||
|
t.Fatalf("want the loudest (service_down/sev4), got %s/sev%d", got.Rule.Name, got.Severity)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Tick returns a single Candidate by type, so "one per tick" cannot be violated
|
||||||
|
// by count — what can drift is WHICH one. Equal severities tie-break by name so
|
||||||
|
// the choice is deterministic across ticks.
|
||||||
|
func TestTickTieBreaksByNameForDeterminism(t *testing.T) {
|
||||||
|
s := State{Now: refTime(), Presence: store.Present}
|
||||||
|
rules := []Rule{testRule("zebra", Sev2), testRule("apple", Sev2), testRule("mango", Sev2)}
|
||||||
|
for i := 0; i < 5; i++ {
|
||||||
|
got := Tick(s, rules)
|
||||||
|
if got == nil || got.Rule.Name != "apple" {
|
||||||
|
t.Fatalf("tie-break: want apple every time, got %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The loudest candidate wins even when the quiet one is listed first.
|
||||||
|
func TestTickOrderOfRulesDoesNotMatter(t *testing.T) {
|
||||||
|
s := State{Now: refTime(), Presence: store.Present}
|
||||||
|
first := Tick(s, []Rule{testRule("care", Sev1), testRule("ops", Sev4)})
|
||||||
|
second := Tick(s, []Rule{testRule("ops", Sev4), testRule("care", Sev1)})
|
||||||
|
if first == nil || second == nil {
|
||||||
|
t.Fatal("want a candidate from both orderings")
|
||||||
|
}
|
||||||
|
if first.Rule.Name != "ops" || second.Rule.Name != "ops" {
|
||||||
|
t.Fatalf("order changed the pick: %s then %s", first.Rule.Name, second.Rule.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- reminders bypass the gate ----------------------
|
||||||
|
|
||||||
|
// DESIGN.md § User reminders: "bypasses the restraint gate — 'wake me 7' fires
|
||||||
|
// in quiet hours; that's the point." Every suppressor set at once, and the
|
||||||
|
// reminder still comes through.
|
||||||
|
func TestRemindersBypassEverySuppressor(t *testing.T) {
|
||||||
|
now := refTime()
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Away,
|
||||||
|
QuietHours: true,
|
||||||
|
CalendarBusy: true,
|
||||||
|
CooldownUntil: map[string]time.Time{"reminder": now.Add(time.Hour)},
|
||||||
|
}
|
||||||
|
due := []store.Reminder{{ID: 7, Payload: `{"text":"wake me"}`}}
|
||||||
|
got := RemindDecisions(s, due)
|
||||||
|
if len(got) != 1 || got[0].Reminder.ID != 7 {
|
||||||
|
t.Fatalf("reminder must bypass the gate, got %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// GAP — DESIGN.md § User reminders ends "Snooze still applies." RemindDecisions
|
||||||
|
// passes every due reminder straight through with no snooze check, so a snoozed
|
||||||
|
// reminder fires anyway. The test below is what the contract asks for.
|
||||||
|
func TestRemindersStillHonourSnooze(t *testing.T) {
|
||||||
|
|
||||||
|
now := refTime()
|
||||||
|
s := State{
|
||||||
|
Now: now,
|
||||||
|
Presence: store.Present,
|
||||||
|
SnoozeUntil: map[string]time.Time{"reminder:7": now.Add(time.Hour)},
|
||||||
|
}
|
||||||
|
due := []store.Reminder{{ID: 7, Payload: `{"text":"wake me"}`}}
|
||||||
|
if got := RemindDecisions(s, due); len(got) != 0 {
|
||||||
|
t.Fatalf("snoozed reminder should not be delivered, got %+v", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- digest eligibility ------------------------------
|
||||||
|
|
||||||
|
// A Sev2 care candidate (break) suppressed for a genuine restraint reason is
|
||||||
|
// worth resurfacing later.
|
||||||
|
func TestDigestEligibleSev2SuppressedByRestraint(t *testing.T) {
|
||||||
|
for _, reason := range []string{"quiet_hours", "calendar_busy", "presence"} {
|
||||||
|
if !DigestEligible(Sev2, reason) {
|
||||||
|
t.Errorf("sev2 blocked by %q: want digest-eligible", reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A Sev1 care candidate (water/meal) never digests — a biological timer
|
||||||
|
// nudge is stale by the time anyone could resurface it, so it just drops.
|
||||||
|
func TestDigestEligibleSev1NeverDigests(t *testing.T) {
|
||||||
|
for _, reason := range []string{"quiet_hours", "calendar_busy", "presence"} {
|
||||||
|
if DigestEligible(Sev1, reason) {
|
||||||
|
t.Errorf("sev1 blocked by %q: want drop, got digest-eligible", reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Ops severities are never blocked by these reasons in practice (Gate only
|
||||||
|
// applies quiet_hours/calendar_busy/presence to care severities), but the
|
||||||
|
// boundary itself must refuse to digest a high severity even if asked —
|
||||||
|
// alarms bypass the gate and deliver now, unchanged, never delayed.
|
||||||
|
func TestDigestEligibleNeverDigestsHighSeverity(t *testing.T) {
|
||||||
|
for _, sev := range []Severity{Sev3, Sev4} {
|
||||||
|
for _, reason := range []string{"quiet_hours", "calendar_busy", "presence"} {
|
||||||
|
if DigestEligible(sev, reason) {
|
||||||
|
t.Errorf("sev%d blocked by %q: high severity must never digest", sev, reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// cooldown and snooze are not "suppression" in the digest sense — cooldown
|
||||||
|
// means it was already said recently, snooze means the user asked to not
|
||||||
|
// hear about it. Neither should resurface later just because the severity
|
||||||
|
// matches.
|
||||||
|
func TestDigestEligibleExcludesCooldownAndSnooze(t *testing.T) {
|
||||||
|
for _, reason := range []string{"cooldown", "snooze", "inert_no_data", "predicate", ""} {
|
||||||
|
if DigestEligible(Sev2, reason) {
|
||||||
|
t.Errorf("sev2 blocked by %q: should not be digest-eligible", reason)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// GAP — the gate reads State.SnoozeUntil, but the Gatherer hard-codes it to nil
|
||||||
|
// (internal/loop/gather.go:153), so snooze is dead in the running daemon: the
|
||||||
|
// unit tests above pass while nothing can ever populate the map. This asserts
|
||||||
|
// the Gatherer actually produces a snooze map.
|
||||||
|
func TestGathererPopulatesSnoozeUntil(t *testing.T) {
|
||||||
|
|
||||||
|
ctx := context.Background()
|
||||||
|
st, err := store.Open(ctx, t.TempDir()+"/m.db")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
defer st.Close()
|
||||||
|
|
||||||
|
g := NewGatherer(st, DefaultRules())
|
||||||
|
snap, _, err := g.GatherState(ctx, refTime())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if snap.SnoozeUntil == nil {
|
||||||
|
t.Fatal("Gatherer returned a nil SnoozeUntil map")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -119,6 +119,14 @@ func (g *Gatherer) GatherState(ctx context.Context, now time.Time) (State, []sto
|
|||||||
return State{}, nil, err
|
return State{}, nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// live snoozes — "leave me alone until X", per rule. The `snoozed` outcome
|
||||||
|
// on the nudges table is the whole record; the store turns it into an
|
||||||
|
// expiry. Absent rules mean "not snoozed", which is what the gate reads.
|
||||||
|
snoozeUntil, err := g.store.SnoozedUntil(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
return State{}, nil, err
|
||||||
|
}
|
||||||
|
|
||||||
// env flags — QuietHours / CalendarBusy as config facts.
|
// env flags — QuietHours / CalendarBusy as config facts.
|
||||||
// QuietHours: presence != reachability, sleep/quiet-hours handled separately
|
// QuietHours: presence != reachability, sleep/quiet-hours handled separately
|
||||||
// in the gate. We read a config `quiet_hours` fact for the boolean.
|
// in the gate. We read a config `quiet_hours` fact for the boolean.
|
||||||
@@ -150,7 +158,7 @@ func (g *Gatherer) GatherState(ctx context.Context, now time.Time) (State, []sto
|
|||||||
PresenceScore: score,
|
PresenceScore: score,
|
||||||
Facts: facts,
|
Facts: facts,
|
||||||
LastNudge: lastNudge,
|
LastNudge: lastNudge,
|
||||||
SnoozeUntil: nil, // no snooze persistence yet — daemon wires in
|
SnoozeUntil: snoozeUntil,
|
||||||
CooldownUntil: cooldownUntil,
|
CooldownUntil: cooldownUntil,
|
||||||
QuietHours: quiet,
|
QuietHours: quiet,
|
||||||
CalendarBusy: calBusy,
|
CalendarBusy: calBusy,
|
||||||
|
|||||||
+65
-6
@@ -1,6 +1,7 @@
|
|||||||
package loop
|
package loop
|
||||||
|
|
||||||
import (
|
import (
|
||||||
|
"fmt"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
"github.com/kami/maven/internal/store"
|
"github.com/kami/maven/internal/store"
|
||||||
@@ -104,27 +105,85 @@ func Tick(s State, rules []Rule) *Candidate {
|
|||||||
return fire
|
return fire
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// DigestEligible decides digest-vs-drop for a care candidate the gate
|
||||||
|
// suppressed this tick (see ExplainGate's blockedBy). Pure — no I/O, no
|
||||||
|
// state, just the two facts that matter: why it was suppressed, and how
|
||||||
|
// insistent it was.
|
||||||
|
//
|
||||||
|
// Only genuine RESTRAINT blocks are eligible at all — quiet_hours,
|
||||||
|
// calendar_busy, presence(away). cooldown and snooze are not suppression in
|
||||||
|
// this sense: cooldown means "you already heard this recently" (resurfacing
|
||||||
|
// it later would be an actual repeat, not a rescue) and snooze is the user
|
||||||
|
// explicitly saying "not this" (digesting it anyway would defeat the ask).
|
||||||
|
// Ops severities (Sev3/4) never reach here — the gate never blocks them for
|
||||||
|
// these reasons in the first place (see Gate), and even if a future rule
|
||||||
|
// dropped Sev3+ into "care", digest still refuses them: alarms bypass the
|
||||||
|
// gate on purpose and must never be silently delayed into a bundle.
|
||||||
|
//
|
||||||
|
// Within care (Sev1–2), the boundary is severity itself: Sev1 (water, meal —
|
||||||
|
// biological timers with no "still relevant later" property; a water nudge
|
||||||
|
// from 3 hours into quiet hours is just wrong by morning) drops. Sev2
|
||||||
|
// (break — "you worked through a long stretch without a break while I
|
||||||
|
// couldn't reach you") is information that stays true and useful after the
|
||||||
|
// fact, so it digests.
|
||||||
|
func DigestEligible(sev Severity, blockedBy string) bool {
|
||||||
|
switch blockedBy {
|
||||||
|
case "quiet_hours", "calendar_busy", "presence":
|
||||||
|
default:
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return sev == Sev2
|
||||||
|
}
|
||||||
|
|
||||||
// ReminderDecision — a due reminder the daemon should deliver now.
|
// ReminderDecision — a due reminder the daemon should deliver now.
|
||||||
// NOT gated by the universal Gate (per spec: "wake me 7" fires in quiet hours;
|
// NOT gated by the universal Gate (per spec: "wake me 7" fires in quiet hours;
|
||||||
// that's the point). Snooze still applies — represented by a separate
|
// that's the point). Snooze is the one part of restraint that still applies.
|
||||||
// snooze-until the gatherer consults; for the scaffold, fired-reminders move
|
|
||||||
// straight to MarkReminder(fired).
|
|
||||||
type ReminderDecision struct {
|
type ReminderDecision struct {
|
||||||
Reminder store.Reminder
|
Reminder store.Reminder
|
||||||
State State
|
State State
|
||||||
}
|
}
|
||||||
|
|
||||||
// RemindDecisions — returns all due reminders (without gating their delivery
|
// ReminderSnoozeKey — the SnoozeUntil key that holds back every due reminder.
|
||||||
// by restraint). Pure: accepts an already-filtered (due) list. The Gatherer
|
// Reminders have no rule name, so they share one key. A snooze aimed at a
|
||||||
// produces that list from `fire_ts <= now AND pending`.
|
// single reminder uses ReminderSnoozeKeyFor instead.
|
||||||
|
const ReminderSnoozeKey = "reminder"
|
||||||
|
|
||||||
|
// ReminderSnoozeKeyFor — the SnoozeUntil key for one reminder by id.
|
||||||
|
func ReminderSnoozeKeyFor(id int64) string {
|
||||||
|
return fmt.Sprintf("%s:%d", ReminderSnoozeKey, id)
|
||||||
|
}
|
||||||
|
|
||||||
|
// RemindDecisions — returns the due reminders the daemon should deliver.
|
||||||
|
// Pure: accepts an already-filtered (due) list. The Gatherer produces that
|
||||||
|
// list from `fire_ts <= now AND pending`.
|
||||||
|
//
|
||||||
|
// Quiet hours, presence and cooldown are deliberately NOT consulted — a
|
||||||
|
// reminder must wake you at 7 even in the middle of quiet hours. Only snooze
|
||||||
|
// holds one back. A held reminder stays pending, so it comes back once the
|
||||||
|
// snooze runs out.
|
||||||
func RemindDecisions(s State, due []store.Reminder) []ReminderDecision {
|
func RemindDecisions(s State, due []store.Reminder) []ReminderDecision {
|
||||||
out := make([]ReminderDecision, 0, len(due))
|
out := make([]ReminderDecision, 0, len(due))
|
||||||
for _, r := range due {
|
for _, r := range due {
|
||||||
|
if reminderSnoozed(s, r) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
out = append(out, ReminderDecision{Reminder: r, State: s})
|
out = append(out, ReminderDecision{Reminder: r, State: s})
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// reminderSnoozed — true when a snooze on this reminder, or on reminders as a
|
||||||
|
// class, is still running.
|
||||||
|
func reminderSnoozed(s State, r store.Reminder) bool {
|
||||||
|
keys := []string{ReminderSnoozeKey, ReminderSnoozeKeyFor(r.ID)}
|
||||||
|
for _, k := range keys {
|
||||||
|
if until, ok := s.SnoozeUntil[k]; ok && s.Now.Before(until) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
// CooldownFor — helper for the Gatherer: given the active cooldown base
|
// CooldownFor — helper for the Gatherer: given the active cooldown base
|
||||||
// (the rule's static Base, OR the feedback tuner's persisted tuning) and the
|
// (the rule's static Base, OR the feedback tuner's persisted tuning) and the
|
||||||
// last send ts, compute the wall-clock "cooldown-until" the gate will check.
|
// last send ts, compute the wall-clock "cooldown-until" the gate will check.
|
||||||
|
|||||||
@@ -0,0 +1,394 @@
|
|||||||
|
package loop
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Direct tests for the five default rule predicates.
|
||||||
|
//
|
||||||
|
// A predicate is pure — (State) -> bool, no I/O — so these need no store and no
|
||||||
|
// daemon. They test the predicate ALONE: the restraint gate is tested in
|
||||||
|
// loop_test.go and gate_test.go, never here.
|
||||||
|
//
|
||||||
|
// Every rule gets the same three questions plus its own edges:
|
||||||
|
// - does it fire when it should?
|
||||||
|
// - does it stay quiet when it should?
|
||||||
|
// - is it silent when the key it needs has no data at all?
|
||||||
|
//
|
||||||
|
// The last one is load-bearing. DESIGN.md: "since(key)==null → don't fire.
|
||||||
|
// Silence on no-data is 'shuts up when uncertain'."
|
||||||
|
|
||||||
|
// stateWith builds a snapshot at refTime() holding just the given facts.
|
||||||
|
// Presence and the env flags are left zero — the predicate must not read them.
|
||||||
|
func stateWith(facts map[string]store.Fact) State {
|
||||||
|
return State{Now: refTime(), Facts: facts}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ago is a fact for key written `d` before refTime().
|
||||||
|
func ago(key, source, value string, d time.Duration) store.Fact {
|
||||||
|
return factAt(key, source, value, refTime().Add(-d))
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- since-based care rules -------------------------
|
||||||
|
|
||||||
|
// The three care rules share one shape: "fire when it has been at least N since
|
||||||
|
// the last fact for key". One table drives all of them.
|
||||||
|
func TestCareRulePredicates(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
rule Rule
|
||||||
|
facts map[string]store.Fact
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
// water — threshold 3h.
|
||||||
|
{
|
||||||
|
name: "water fires at 4h",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: map[string]store.Fact{"water": ago("water", "tap:water", `"250ml"`, 4*time.Hour)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "water fires exactly at the 3h threshold",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: map[string]store.Fact{"water": ago("water", "tap:water", `"250ml"`, 3*time.Hour)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "water quiet just under 3h",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: map[string]store.Fact{"water": ago("water", "tap:water", `"250ml"`, 3*time.Hour-time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "water quiet on no data",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: nil,
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "water quiet on a zero-timestamp fact",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: map[string]store.Fact{"water": {Key: "water", Source: "tap:water", Value: `"250ml"`}},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "water quiet when the only fact is for another key",
|
||||||
|
rule: WaterRule(),
|
||||||
|
facts: map[string]store.Fact{"meal": ago("meal", "voice", `"lunch"`, 9*time.Hour)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
|
||||||
|
// meal — threshold 6h.
|
||||||
|
{
|
||||||
|
name: "meal fires at 7h",
|
||||||
|
rule: MealRule(),
|
||||||
|
facts: map[string]store.Fact{"meal": ago("meal", "voice", `"lunch"`, 7*time.Hour)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "meal fires exactly at the 6h threshold",
|
||||||
|
rule: MealRule(),
|
||||||
|
facts: map[string]store.Fact{"meal": ago("meal", "voice", `"lunch"`, 6*time.Hour)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "meal quiet just under 6h",
|
||||||
|
rule: MealRule(),
|
||||||
|
facts: map[string]store.Fact{"meal": ago("meal", "voice", `"lunch"`, 6*time.Hour-time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "meal quiet on no data",
|
||||||
|
rule: MealRule(),
|
||||||
|
facts: nil,
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
|
||||||
|
// break — needs BOTH anchors: at the desk now, and no break for 90min.
|
||||||
|
{
|
||||||
|
name: "break fires when at desk and no break for 2h",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 30*time.Second),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 2*time.Hour),
|
||||||
|
},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break fires exactly at both thresholds",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 2*time.Minute),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 90*time.Minute),
|
||||||
|
},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break quiet when the desk signal is stale (user left)",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 10*time.Minute),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 2*time.Hour),
|
||||||
|
},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break quiet when the last break was recent",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 30*time.Second),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 20*time.Minute),
|
||||||
|
},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break quiet with only the desk anchor",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 30*time.Second),
|
||||||
|
},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break quiet with only the break anchor",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: map[string]store.Fact{
|
||||||
|
"break": ago("break", "voice", `"walk"`, 2*time.Hour),
|
||||||
|
},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "break quiet on no data",
|
||||||
|
rule: BreakRule(),
|
||||||
|
facts: nil,
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := c.rule.Predicate(stateWith(c.facts)); got != c.want {
|
||||||
|
t.Fatalf("%s predicate: want %v, got %v", c.rule.Name, c.want, got)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- ops rules --------------------------------------
|
||||||
|
|
||||||
|
// The two ops rules match on a value AND on which poller wrote it. DESIGN.md:
|
||||||
|
// "a compromised poller must not be able to forge a trigger." Half of this
|
||||||
|
// table is forgery attempts; all of them must be refused.
|
||||||
|
func TestOpsRulePredicates(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
rule Rule
|
||||||
|
facts map[string]store.Fact
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
// service_down — only poll:uptimekuma may say a service is down.
|
||||||
|
{
|
||||||
|
name: "service_down fires on a kuma down fact",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "poll:uptimekuma", `"down"`, time.Minute)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down quiet when kuma says up",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "poll:uptimekuma", `"up"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down quiet on no data",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: nil,
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down quiet on a zero-timestamp fact",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": {Key: "service_down", Source: "poll:uptimekuma", Value: `"down"`}},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
// forgery attempts — right value, wrong writer.
|
||||||
|
{
|
||||||
|
name: "service_down refuses a forgery from the netdata poller",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "poll:netdata", `"down"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down refuses a forgery from ambient audio",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "ambient:other", `"down"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down refuses a forgery from the user's own voice",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "voice", `"down"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down refuses a source that only looks like kuma",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "poll:uptimekuma-staging", `"down"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "service_down refuses an unquoted down value",
|
||||||
|
rule: ServiceDownRule(),
|
||||||
|
facts: map[string]store.Fact{"service_down": ago("service_down", "poll:uptimekuma", `down`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
|
||||||
|
// netdata_critical — only poll:netdata may raise a critical alarm.
|
||||||
|
{
|
||||||
|
name: "netdata_critical fires on a netdata critical alarm",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": ago("netdata_alarm", "poll:netdata", `"critical"`, time.Minute)},
|
||||||
|
want: true,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical quiet on a warning alarm",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": ago("netdata_alarm", "poll:netdata", `"warning"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical quiet on a cleared alarm",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": ago("netdata_alarm", "poll:netdata", `"clear"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical quiet on no data",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: nil,
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical quiet on a zero-timestamp fact",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": {Key: "netdata_alarm", Source: "poll:netdata", Value: `"critical"`}},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical refuses a forgery from the kuma poller",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": ago("netdata_alarm", "poll:uptimekuma", `"critical"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical refuses a forgery from ambient audio",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_alarm": ago("netdata_alarm", "ambient:other", `"critical"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "netdata_critical reads netdata_alarm, not netdata_critical",
|
||||||
|
rule: NetdataCriticalRule(),
|
||||||
|
facts: map[string]store.Fact{"netdata_critical": ago("netdata_critical", "poll:netdata", `"critical"`, time.Minute)},
|
||||||
|
want: false,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := c.rule.Predicate(stateWith(c.facts)); got != c.want {
|
||||||
|
t.Fatalf("%s predicate: want %v, got %v", c.rule.Name, c.want, got)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------- rule metadata ----------------------------------
|
||||||
|
|
||||||
|
// Every default rule must declare the keys it needs. The gate uses that list as
|
||||||
|
// a second no-data backstop, so a rule that forgets it loses the safety net
|
||||||
|
// even if its predicate happens to check.
|
||||||
|
func TestDefaultRulesDeclareInertKeys(t *testing.T) {
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
if len(r.InertWhenNoData) == 0 {
|
||||||
|
t.Errorf("rule %q declares no InertWhenNoData keys", r.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A no-data snapshot must make EVERY default rule quiet, predicate alone, with
|
||||||
|
// the gate out of the picture. This is the whole-set version of the per-rule
|
||||||
|
// no-data cases above.
|
||||||
|
func TestNoDefaultRuleFiresOnEmptyState(t *testing.T) {
|
||||||
|
empty := stateWith(nil)
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
if r.Predicate(empty) {
|
||||||
|
t.Errorf("rule %q fires on an empty snapshot", r.Name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Severities are the delivery contract (DESIGN.md § Delivery / channel
|
||||||
|
// routing): care is sev1-2 and drops when away, ops is sev3-4 and holds. Pin
|
||||||
|
// them so a change to a rule's insistence has to be deliberate.
|
||||||
|
func TestDefaultRuleSeverities(t *testing.T) {
|
||||||
|
want := map[string]Severity{
|
||||||
|
"water": Sev1,
|
||||||
|
"meal": Sev1,
|
||||||
|
"break": Sev2,
|
||||||
|
"service_down": Sev4,
|
||||||
|
"netdata_critical": Sev3,
|
||||||
|
}
|
||||||
|
got := map[string]Severity{}
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
got[r.Name] = r.Severity
|
||||||
|
}
|
||||||
|
if len(got) != len(want) {
|
||||||
|
t.Fatalf("rule count changed: want %d, got %d", len(want), len(got))
|
||||||
|
}
|
||||||
|
for name, sev := range want {
|
||||||
|
if got[name] != sev {
|
||||||
|
t.Errorf("rule %q severity: want %d, got %d", name, sev, got[name])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cooldown bounds keep the feedback tuner honest — DESIGN.md wants
|
||||||
|
// `cooldown in [min,max]` "so a weird week can't mutate Maven silent or
|
||||||
|
// stalker". A base outside its own envelope would make that meaningless.
|
||||||
|
func TestDefaultRuleCooldownsAreBounded(t *testing.T) {
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
c := r.Cooldown
|
||||||
|
if c.Min <= 0 || c.Base <= 0 || c.Max <= 0 {
|
||||||
|
t.Errorf("rule %q has a non-positive cooldown: %+v", r.Name, c)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if c.Base < c.Min || c.Base > c.Max {
|
||||||
|
t.Errorf("rule %q base %v outside envelope [%v, %v]", r.Name, c.Base, c.Min, c.Max)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A predicate must read only the snapshot it is handed. Same snapshot twice
|
||||||
|
// (and a snapshot shared between two rules) must give the same answer — no
|
||||||
|
// hidden state, no clock reads.
|
||||||
|
func TestPredicatesArePure(t *testing.T) {
|
||||||
|
s := stateWith(map[string]store.Fact{
|
||||||
|
"water": ago("water", "tap:water", `"250ml"`, 4*time.Hour),
|
||||||
|
"meal": ago("meal", "voice", `"lunch"`, 7*time.Hour),
|
||||||
|
"desk_active": ago("desk_active", "infer:hyprland", "1", 30*time.Second),
|
||||||
|
"break": ago("break", "voice", `"walk"`, 2*time.Hour),
|
||||||
|
"service_down": ago("service_down", "poll:uptimekuma", `"down"`, time.Minute),
|
||||||
|
})
|
||||||
|
for _, r := range DefaultRules() {
|
||||||
|
first := r.Predicate(s)
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
if again := r.Predicate(s); again != first {
|
||||||
|
t.Fatalf("rule %q predicate is not pure: %v then %v", r.Name, first, again)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,357 @@
|
|||||||
|
// Package memeval is background memory evaluation (Vikunja #248,
|
||||||
|
// docs/plans/03-memory-evaluation.md).
|
||||||
|
//
|
||||||
|
// It lives beside internal/memory rather than inside it because
|
||||||
|
// internal/store imports internal/memory for the vector-store backend, and an
|
||||||
|
// evaluator has to read store.Fact / store.Note / store.Nudge — putting it in
|
||||||
|
// internal/memory would close that import cycle.
|
||||||
|
//
|
||||||
|
// Every so often Maven reads back her own recent memory — facts, notes, the
|
||||||
|
// nudges she sent — and asks the resident model what it notices: a habit that
|
||||||
|
// stopped, a gap, something worth saying later. What comes back is written as
|
||||||
|
// notes with source EvalNoteSource and nothing else happens. That restraint is
|
||||||
|
// the design, not an unfinished edge:
|
||||||
|
//
|
||||||
|
// - She does not speak here. There is no dispatcher, no channel, no nudge.
|
||||||
|
// An observation is a thought she wrote down; he reads it on /dash when he
|
||||||
|
// wants to. "Not a nag, not autonomous" (CLAUDE.md) is easy to violate with
|
||||||
|
// exactly this feature — an hourly loop with an LLM in it and permission to
|
||||||
|
// talk is a machine for generating interruptions — so the loop has no way
|
||||||
|
// to reach him at all. Turning observations into nudges is a separate
|
||||||
|
// decision with a separate opt-in, and it is deliberately NOT in this file.
|
||||||
|
// - She does not act. No reminder is created, no routine proposed, no fact
|
||||||
|
// written. The model's suggested_action is recorded as text inside the note
|
||||||
|
// and interpreted by nobody.
|
||||||
|
// - She says nothing about an empty store. No memory ⇒ no LLM call ⇒ no
|
||||||
|
// "observations" invented out of two facts. A 1.7B asked to find a pattern
|
||||||
|
// will always find one; the defence is not asking.
|
||||||
|
//
|
||||||
|
// Everything the evaluator writes is attributable: source is EvalNoteSource, so
|
||||||
|
// an inferred observation can never be mistaken for something he said, and the
|
||||||
|
// whole batch is one SQL delete away if the output turns out to be noise.
|
||||||
|
package memeval
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"encoding/json"
|
||||||
|
"fmt"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
"github.com/kami/maven/internal/persona"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// EvalNoteSource — the source stamped on every note the evaluator writes.
|
||||||
|
// Same infer:* convention as the rest of the derived facts.
|
||||||
|
const EvalNoteSource = "infer:memory-eval"
|
||||||
|
|
||||||
|
// DefaultMinConfidence — an observation below this is dropped. The model is
|
||||||
|
// asked for its own confidence and small models are badly calibrated, so this
|
||||||
|
// is a coarse filter, not a probability: it exists to throw away the guesses
|
||||||
|
// the model itself hedged on.
|
||||||
|
const DefaultMinConfidence = 0.7
|
||||||
|
|
||||||
|
// DefaultMaxItems — how much recent memory goes into one evaluation, per
|
||||||
|
// store. 30 facts + 30 notes + 30 nudges is a few thousand tokens of the 4096
|
||||||
|
// context the resident Thinking model runs with, which leaves room for its
|
||||||
|
// reasoning tokens. Raising this trades reasoning room for history.
|
||||||
|
const DefaultMaxItems = 30
|
||||||
|
|
||||||
|
// MaxObservations — the model may return at most this many observations per
|
||||||
|
// evaluation, enforced by the grammar. A cap here is also a noise cap: an
|
||||||
|
// evaluation that "notices" ten things has noticed nothing.
|
||||||
|
const MaxObservations = 3
|
||||||
|
|
||||||
|
// Observation — one thing the evaluator noticed.
|
||||||
|
type Observation struct {
|
||||||
|
Text string `json:"observation"`
|
||||||
|
Conf float64 `json:"confidence"`
|
||||||
|
// Action — what the model thinks should happen with this. Recorded, never
|
||||||
|
// executed: see the file comment. One of "note", "propose", "notify".
|
||||||
|
Action string `json:"suggested_action"`
|
||||||
|
}
|
||||||
|
|
||||||
|
// Completer — the llama-server seam, same shape router.Completer uses so the
|
||||||
|
// one resident model serves this caller too.
|
||||||
|
type Completer interface {
|
||||||
|
Complete(ctx context.Context, r llm.Req) (string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Reader — the slice of the store an evaluation reads. Narrow on purpose: the
|
||||||
|
// evaluator gets recent memory and nothing else. No entity graph, no presence,
|
||||||
|
// no config facts.
|
||||||
|
type Reader interface {
|
||||||
|
RecentFacts(ctx context.Context, n int) ([]store.Fact, error)
|
||||||
|
RecentNotes(ctx context.Context, n int) ([]store.Note, error)
|
||||||
|
RecentNudges(ctx context.Context, n int) ([]store.Nudge, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// NoteWriter — where observations land. Embeddings are passed nil: an
|
||||||
|
// observation is written for a human to read on /dash, not to be recalled by
|
||||||
|
// similarity. Feeding LLM-generated text back into the RAG pool it was
|
||||||
|
// generated from is how a small model starts citing its own guesses as
|
||||||
|
// evidence.
|
||||||
|
type NoteWriter interface {
|
||||||
|
WriteNote(ctx context.Context, ts time.Time, text string, embedding []float32, source string) (int64, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Config — evaluator tuning. Zero values are replaced by the Default*
|
||||||
|
// constants, so the zero Config is the sane one.
|
||||||
|
type Config struct {
|
||||||
|
MaxItems int
|
||||||
|
MinConfidence float64
|
||||||
|
|
||||||
|
// ContextBlock — the shared persona block (internal/persona), re-evaluated
|
||||||
|
// per call so the clock in it is current. Prepended to the system prompt so
|
||||||
|
// observations come out in Maven's voice: feminine self-reference, informal
|
||||||
|
// "ты". nil is allowed; the base prompt still carries the address rules.
|
||||||
|
ContextBlock func() string
|
||||||
|
}
|
||||||
|
|
||||||
|
// Evaluator reads recent memory and records what the model notices.
|
||||||
|
type Evaluator struct {
|
||||||
|
read Reader
|
||||||
|
write NoteWriter
|
||||||
|
llm Completer
|
||||||
|
cfg Config
|
||||||
|
}
|
||||||
|
|
||||||
|
func NewEvaluator(r Reader, w NoteWriter, c Completer, cfg Config) *Evaluator {
|
||||||
|
if cfg.MaxItems <= 0 {
|
||||||
|
cfg.MaxItems = DefaultMaxItems
|
||||||
|
}
|
||||||
|
if cfg.MinConfidence <= 0 {
|
||||||
|
cfg.MinConfidence = DefaultMinConfidence
|
||||||
|
}
|
||||||
|
return &Evaluator{read: r, write: w, llm: c, cfg: cfg}
|
||||||
|
}
|
||||||
|
|
||||||
|
// evalGrammar — GBNF pinning the reply to a bounded JSON array of fixed-shape
|
||||||
|
// observations. Same reasoning as the router's routeGrammar: the enum and the
|
||||||
|
// length bound are what stop a small model from drifting into free text or
|
||||||
|
// filling the token budget with one repeated field.
|
||||||
|
const evalGrammar = `
|
||||||
|
root ::= "[" ws (obs ("," ws obs){0,2})? ws "]"
|
||||||
|
obs ::= "{" ws "\"observation\"" ws ":" ws text "," ws "\"confidence\"" ws ":" ws conf "," ws "\"suggested_action\"" ws ":" ws act ws "}"
|
||||||
|
text ::= "\"" ([^"\\] | "\\" .){1,200} "\""
|
||||||
|
conf ::= "0" "." [0-9]{1,2} | "1" ("." "0")?
|
||||||
|
act ::= "\"note\"" | "\"propose\"" | "\"notify\""
|
||||||
|
ws ::= [ \t\n]*
|
||||||
|
`
|
||||||
|
|
||||||
|
// evalSystem — the evaluation prompt. Two things it insists on, both learned
|
||||||
|
// from the phraser: state the observation as something she noticed rather than
|
||||||
|
// an instruction, and say nothing when there is nothing (the model is given an
|
||||||
|
// explicit way to return an empty array, because a model with no exit returns
|
||||||
|
// filler).
|
||||||
|
const evalSystem = `Ты просматриваешь свою собственную память: недавние факты, заметки и напоминания, которые ты отправляла.
|
||||||
|
Найди то, что действительно заметно: привычка, которая прервалась; пробел в записях; повторяющаяся закономерность.
|
||||||
|
|
||||||
|
Правила:
|
||||||
|
- Отвечай ТОЛЬКО массивом JSON. Каждый элемент: {"observation": "...", "confidence": 0.0-1.0, "suggested_action": "note"|"propose"|"notify"}.
|
||||||
|
- observation — короткая фраза по-русски о том, что ты заметила. О себе — в женском роде ("я заметила"). К нему — на "ты".
|
||||||
|
- Не выдумывай. Если в памяти нет ничего заметного, верни пустой массив [].
|
||||||
|
- Не давай советов и не приказывай. Ты замечаешь, а не требуешь.
|
||||||
|
- confidence — насколько ты уверена, что это настоящая закономерность, а не совпадение.
|
||||||
|
- Максимум три наблюдения. Лучше одно точное, чем три общих.`
|
||||||
|
|
||||||
|
// Evaluate runs one evaluation and returns the observations it recorded.
|
||||||
|
//
|
||||||
|
// Returns (nil, nil) — not an error — for every ordinary "nothing to say"
|
||||||
|
// outcome: an empty store, an empty array from the model, everything below the
|
||||||
|
// confidence floor, or every observation already recorded earlier. Only a real
|
||||||
|
// read/LLM/write failure is an error, and the caller (a background ticker) logs
|
||||||
|
// it and waits for the next interval.
|
||||||
|
func (e *Evaluator) Evaluate(ctx context.Context, now time.Time) ([]Observation, error) {
|
||||||
|
snap, err := e.snapshot(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if snap == "" {
|
||||||
|
return nil, nil // nothing recorded ⇒ nothing to notice, and no LLM call
|
||||||
|
}
|
||||||
|
|
||||||
|
raw, err := e.llm.Complete(ctx, llm.Req{
|
||||||
|
System: persona.Prepend(e.cfg.ContextBlock, evalSystem),
|
||||||
|
User: snap,
|
||||||
|
Grammar: evalGrammar,
|
||||||
|
MaxTokens: 512,
|
||||||
|
RepeatPenalty: 1.1,
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("memory eval: complete: %w", err)
|
||||||
|
}
|
||||||
|
obs, err := parseObservations(raw)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("memory eval: parse %q: %w", truncate(raw, 120), err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Dedupe against what earlier evaluations already wrote. Without this an
|
||||||
|
// hourly loop over a slowly-changing store writes the same sentence every
|
||||||
|
// hour until /dash is nothing but the evaluator talking to itself.
|
||||||
|
seen, err := e.recordedTexts(ctx)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
var kept []Observation
|
||||||
|
for _, o := range obs {
|
||||||
|
o.Text = strings.TrimSpace(o.Text)
|
||||||
|
if o.Text == "" || o.Conf < e.cfg.MinConfidence {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
norm := normalizeObservation(o.Text)
|
||||||
|
if seen[norm] {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[norm] = true
|
||||||
|
if _, err := e.write.WriteNote(ctx, now, formatNote(o), nil, EvalNoteSource); err != nil {
|
||||||
|
return kept, fmt.Errorf("memory eval: write note: %w", err)
|
||||||
|
}
|
||||||
|
kept = append(kept, o)
|
||||||
|
}
|
||||||
|
return kept, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// formatNote — the stored text. The suggested action is kept as a visible
|
||||||
|
// suffix rather than a column: it is the model's opinion about what to do next,
|
||||||
|
// and the only consumer is a human reading /dash.
|
||||||
|
func formatNote(o Observation) string {
|
||||||
|
if o.Action == "" {
|
||||||
|
return o.Text
|
||||||
|
}
|
||||||
|
return fmt.Sprintf("%s [%s]", o.Text, o.Action)
|
||||||
|
}
|
||||||
|
|
||||||
|
// recordedTexts — the normalized text of every observation earlier evaluations
|
||||||
|
// wrote, for dedupe. Reads a wider window than MaxItems because the point is to
|
||||||
|
// remember saying it, not to summarize it.
|
||||||
|
func (e *Evaluator) recordedTexts(ctx context.Context) (map[string]bool, error) {
|
||||||
|
notes, err := e.read.RecentNotes(ctx, 200)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("memory eval: recent notes: %w", err)
|
||||||
|
}
|
||||||
|
seen := make(map[string]bool, len(notes))
|
||||||
|
for _, n := range notes {
|
||||||
|
if n.Source != EvalNoteSource {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
text := n.Text
|
||||||
|
// Strip the "[action]" suffix formatNote appended.
|
||||||
|
if i := strings.LastIndex(text, " ["); i > 0 && strings.HasSuffix(text, "]") {
|
||||||
|
text = text[:i]
|
||||||
|
}
|
||||||
|
seen[normalizeObservation(text)] = true
|
||||||
|
}
|
||||||
|
return seen, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// normalizeObservation — dedupe key. Case- and whitespace-insensitive, which
|
||||||
|
// catches the realistic repeat (the model re-emitting the same sentence with a
|
||||||
|
// different comma) without pretending to do semantic dedupe.
|
||||||
|
func normalizeObservation(s string) string {
|
||||||
|
return strings.Join(strings.Fields(strings.ToLower(s)), " ")
|
||||||
|
}
|
||||||
|
|
||||||
|
// snapshot renders recent memory as the user turn. Returns "" when there is
|
||||||
|
// nothing in any store — the caller treats that as "do not ask the model".
|
||||||
|
//
|
||||||
|
// Notes written by earlier evaluations are excluded. Feeding her own
|
||||||
|
// observations back in is how "я заметила, что ты не записывал еду" becomes
|
||||||
|
// evidence for noticing it again, three evaluations deep.
|
||||||
|
func (e *Evaluator) snapshot(ctx context.Context) (string, error) {
|
||||||
|
n := e.cfg.MaxItems
|
||||||
|
facts, err := e.read.RecentFacts(ctx, n)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("memory eval: recent facts: %w", err)
|
||||||
|
}
|
||||||
|
notes, err := e.read.RecentNotes(ctx, n)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("memory eval: recent notes: %w", err)
|
||||||
|
}
|
||||||
|
nudges, err := e.read.RecentNudges(ctx, n)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("memory eval: recent nudges: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
wrote := false
|
||||||
|
if len(facts) > 0 {
|
||||||
|
b.WriteString("Факты:\n")
|
||||||
|
for _, f := range facts {
|
||||||
|
fmt.Fprintf(&b, "- %s %s=%s (%s)\n", f.Ts.Format("2006-01-02 15:04"), f.Key, truncate(f.Value, 80), f.Source)
|
||||||
|
wrote = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
own := 0
|
||||||
|
var noteLines []string
|
||||||
|
for _, nt := range notes {
|
||||||
|
if nt.Source == EvalNoteSource {
|
||||||
|
own++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
noteLines = append(noteLines, fmt.Sprintf("- %s %s\n", nt.Ts.Format("2006-01-02 15:04"), truncate(nt.Text, 160)))
|
||||||
|
}
|
||||||
|
if len(noteLines) > 0 {
|
||||||
|
b.WriteString("\nЗаметки:\n")
|
||||||
|
for _, l := range noteLines {
|
||||||
|
b.WriteString(l)
|
||||||
|
wrote = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(nudges) > 0 {
|
||||||
|
b.WriteString("\nНапоминания, которые ты отправляла:\n")
|
||||||
|
for _, nd := range nudges {
|
||||||
|
outcome := nd.Outcome
|
||||||
|
if outcome == "" {
|
||||||
|
outcome = "?"
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, "- %s %s → %s (%s)\n", nd.Ts.Format("2006-01-02 15:04"), nd.Rule, outcome, nd.Channel)
|
||||||
|
wrote = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !wrote {
|
||||||
|
// Only her own past observations, or nothing at all. Either way there is
|
||||||
|
// no new memory to evaluate.
|
||||||
|
return "", nil
|
||||||
|
}
|
||||||
|
b.WriteString("\nЧто ты замечаешь?")
|
||||||
|
return b.String(), nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// parseObservations reads the model's array. Tolerates the leading/trailing
|
||||||
|
// prose a Thinking model sometimes emits around JSON by taking the outermost
|
||||||
|
// bracketed span, the same tolerance the router's parser has.
|
||||||
|
func parseObservations(raw string) ([]Observation, error) {
|
||||||
|
s := strings.TrimSpace(raw)
|
||||||
|
if i := strings.Index(s, "["); i >= 0 {
|
||||||
|
if j := strings.LastIndex(s, "]"); j > i {
|
||||||
|
s = s[i : j+1]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if s == "" {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
var obs []Observation
|
||||||
|
if err := json.Unmarshal([]byte(s), &obs); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if len(obs) > MaxObservations {
|
||||||
|
// The grammar bounds this; a grammar-less server or a future prompt
|
||||||
|
// change must not be able to flood /dash.
|
||||||
|
sort.SliceStable(obs, func(i, j int) bool { return obs[i].Conf > obs[j].Conf })
|
||||||
|
obs = obs[:MaxObservations]
|
||||||
|
}
|
||||||
|
return obs, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func truncate(s string, n int) string {
|
||||||
|
r := []rune(s)
|
||||||
|
if len(r) <= n {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return string(r[:n]) + "…"
|
||||||
|
}
|
||||||
@@ -0,0 +1,271 @@
|
|||||||
|
package memeval
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"database/sql"
|
||||||
|
"errors"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kami/maven/internal/llm"
|
||||||
|
"github.com/kami/maven/internal/store"
|
||||||
|
)
|
||||||
|
|
||||||
|
// fakeLLM — canned replies, one per call, and a record of what it was asked.
|
||||||
|
type fakeLLM struct {
|
||||||
|
replies []string
|
||||||
|
calls []llm.Req
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *fakeLLM) Complete(_ context.Context, r llm.Req) (string, error) {
|
||||||
|
f.calls = append(f.calls, r)
|
||||||
|
if f.err != nil {
|
||||||
|
return "", f.err
|
||||||
|
}
|
||||||
|
if len(f.replies) == 0 {
|
||||||
|
return "[]", nil
|
||||||
|
}
|
||||||
|
out := f.replies[0]
|
||||||
|
f.replies = f.replies[1:]
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func newTestStore(t *testing.T) *store.Store {
|
||||||
|
t.Helper()
|
||||||
|
st, err := store.Open(context.Background(), filepath.Join(t.TempDir(), "memeval_test.db"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("store.Open: %v", err)
|
||||||
|
}
|
||||||
|
t.Cleanup(func() { _ = st.Close() })
|
||||||
|
return st
|
||||||
|
}
|
||||||
|
|
||||||
|
func refNow() time.Time { return time.Date(2026, 8, 1, 9, 0, 0, 0, time.UTC) }
|
||||||
|
|
||||||
|
// seedMemory writes a little of everything the evaluator reads.
|
||||||
|
func seedMemory(t *testing.T, st *store.Store, ctx context.Context, now time.Time) {
|
||||||
|
t.Helper()
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
ts := now.Add(-time.Duration(i+1) * 24 * time.Hour)
|
||||||
|
if _, err := st.WriteFact(ctx, ts, store.KindSelf, "water_ml", "500", "tap:desk", 1.0, sql.NullInt64{}); err != nil {
|
||||||
|
t.Fatalf("write fact: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if _, err := st.WriteNote(ctx, now.Add(-2*time.Hour), "купить корм для кота", nil, "tap:voice"); err != nil {
|
||||||
|
t.Fatalf("write note: %v", err)
|
||||||
|
}
|
||||||
|
if _, err := st.RecordNudge(ctx, "water", "voice", "пора выпить воды", now.Add(-time.Hour)); err != nil {
|
||||||
|
t.Fatalf("record nudge: %v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateEmptyStoreAsksNothing — the "shuts up when uncertain" floor. An
|
||||||
|
// empty store must not even reach the model: a small model asked to find a
|
||||||
|
// pattern in nothing will invent one.
|
||||||
|
func TestEvaluateEmptyStoreAsksNothing(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
f := &fakeLLM{}
|
||||||
|
ev := NewEvaluator(st, st, f, Config{})
|
||||||
|
|
||||||
|
obs, err := ev.Evaluate(ctx, refNow())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Evaluate: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != 0 {
|
||||||
|
t.Fatalf("observations on an empty store = %d, want 0", len(obs))
|
||||||
|
}
|
||||||
|
if len(f.calls) != 0 {
|
||||||
|
t.Fatalf("LLM called %d times on an empty store, want 0", len(f.calls))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateWritesHighConfidenceObservations — the happy path. Confident
|
||||||
|
// observations are written as notes stamped infer:memory-eval, and the low
|
||||||
|
// ones are dropped.
|
||||||
|
func TestEvaluateWritesHighConfidenceObservations(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedMemory(t, st, ctx, now)
|
||||||
|
|
||||||
|
f := &fakeLLM{replies: []string{`[
|
||||||
|
{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"notify"},
|
||||||
|
{"observation":"может быть, ты стал меньше пить воды","confidence":0.3,"suggested_action":"note"}
|
||||||
|
]`}}
|
||||||
|
ev := NewEvaluator(st, st, f, Config{})
|
||||||
|
|
||||||
|
obs, err := ev.Evaluate(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Evaluate: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != 1 {
|
||||||
|
t.Fatalf("kept %d observations, want 1 (the 0.3 one is below the floor): %+v", len(obs), obs)
|
||||||
|
}
|
||||||
|
if obs[0].Text != "ты три дня не записывал еду" {
|
||||||
|
t.Errorf("kept the wrong observation: %q", obs[0].Text)
|
||||||
|
}
|
||||||
|
|
||||||
|
notes, err := st.RecentNotes(ctx, 50)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("RecentNotes: %v", err)
|
||||||
|
}
|
||||||
|
var written []store.Note
|
||||||
|
for _, n := range notes {
|
||||||
|
if n.Source == EvalNoteSource {
|
||||||
|
written = append(written, n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(written) != 1 {
|
||||||
|
t.Fatalf("notes with source %s = %d, want 1", EvalNoteSource, len(written))
|
||||||
|
}
|
||||||
|
if !strings.Contains(written[0].Text, "ты три дня не записывал еду") {
|
||||||
|
t.Errorf("note text = %q", written[0].Text)
|
||||||
|
}
|
||||||
|
if !strings.Contains(written[0].Text, "[notify]") {
|
||||||
|
t.Errorf("note text = %q, want the suggested action recorded", written[0].Text)
|
||||||
|
}
|
||||||
|
|
||||||
|
// The prompt must carry the memory it is evaluating, and must not carry a
|
||||||
|
// grammar-free request.
|
||||||
|
if len(f.calls) != 1 {
|
||||||
|
t.Fatalf("LLM calls = %d, want 1", len(f.calls))
|
||||||
|
}
|
||||||
|
if !strings.Contains(f.calls[0].User, "water_ml") {
|
||||||
|
t.Errorf("prompt does not mention the seeded facts:\n%s", f.calls[0].User)
|
||||||
|
}
|
||||||
|
if f.calls[0].Grammar == "" {
|
||||||
|
t.Error("evaluation ran without a grammar")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateDeduplicatesAcrossRuns — the failure mode that would make this
|
||||||
|
// feature unusable: an hourly loop over a store that barely changes writing the
|
||||||
|
// same sentence every hour until /dash is nothing but the evaluator.
|
||||||
|
func TestEvaluateDeduplicatesAcrossRuns(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedMemory(t, st, ctx, now)
|
||||||
|
|
||||||
|
same := `[{"observation":"ты три дня не записывал еду","confidence":0.9,"suggested_action":"note"}]`
|
||||||
|
spaced := `[{"observation":"Ты три дня не записывал еду","confidence":0.95,"suggested_action":"note"}]`
|
||||||
|
f := &fakeLLM{replies: []string{same, same, spaced}}
|
||||||
|
ev := NewEvaluator(st, st, f, Config{})
|
||||||
|
|
||||||
|
for i := 0; i < 3; i++ {
|
||||||
|
if _, err := ev.Evaluate(ctx, now.Add(time.Duration(i)*time.Hour)); err != nil {
|
||||||
|
t.Fatalf("Evaluate %d: %v", i, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
notes, err := st.RecentNotes(ctx, 50)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("RecentNotes: %v", err)
|
||||||
|
}
|
||||||
|
n := 0
|
||||||
|
for _, nt := range notes {
|
||||||
|
if nt.Source == EvalNoteSource {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if n != 1 {
|
||||||
|
t.Fatalf("eval notes after three identical evaluations = %d, want 1", n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateIgnoresOwnNotes — her own observations must not become input.
|
||||||
|
// Otherwise "я заметила X" is evidence for noticing X again, three evaluations
|
||||||
|
// deep. With nothing but eval notes in the store there is no new memory, so the
|
||||||
|
// model is not asked at all.
|
||||||
|
func TestEvaluateIgnoresOwnNotes(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
if _, err := st.WriteNote(ctx, now.Add(-time.Hour), "я заметила, что ты мало пьёшь [note]", nil, EvalNoteSource); err != nil {
|
||||||
|
t.Fatalf("write note: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
f := &fakeLLM{}
|
||||||
|
ev := NewEvaluator(st, st, f, Config{})
|
||||||
|
obs, err := ev.Evaluate(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Evaluate: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != 0 || len(f.calls) != 0 {
|
||||||
|
t.Fatalf("observations=%d llm calls=%d, want 0/0 — own notes are not memory to evaluate", len(obs), len(f.calls))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateEmptyArrayIsNotAnError — "nothing to say" is the expected outcome
|
||||||
|
// most of the time and must not be logged as a failure.
|
||||||
|
func TestEvaluateEmptyArrayIsNotAnError(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedMemory(t, st, ctx, now)
|
||||||
|
|
||||||
|
ev := NewEvaluator(st, st, &fakeLLM{replies: []string{"[]"}}, Config{})
|
||||||
|
obs, err := ev.Evaluate(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Evaluate: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != 0 {
|
||||||
|
t.Fatalf("observations = %d, want 0", len(obs))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestEvaluateLLMErrorIsReported — a broken llama-server is an error the caller
|
||||||
|
// logs; it must not silently write anything.
|
||||||
|
func TestEvaluateLLMErrorIsReported(t *testing.T) {
|
||||||
|
st := newTestStore(t)
|
||||||
|
ctx := context.Background()
|
||||||
|
now := refNow()
|
||||||
|
seedMemory(t, st, ctx, now)
|
||||||
|
|
||||||
|
ev := NewEvaluator(st, st, &fakeLLM{err: errors.New("connection refused")}, Config{})
|
||||||
|
if _, err := ev.Evaluate(ctx, now); err == nil {
|
||||||
|
t.Fatal("want an error when the model is unreachable")
|
||||||
|
}
|
||||||
|
notes, err := st.RecentNotes(ctx, 50)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("RecentNotes: %v", err)
|
||||||
|
}
|
||||||
|
for _, n := range notes {
|
||||||
|
if n.Source == EvalNoteSource {
|
||||||
|
t.Fatalf("wrote a note despite an LLM failure: %q", n.Text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestParseObservationsTolerantAndBounded — Thinking models wrap JSON in prose,
|
||||||
|
// and no reply may exceed MaxObservations even if the grammar is bypassed.
|
||||||
|
func TestParseObservationsTolerantAndBounded(t *testing.T) {
|
||||||
|
obs, err := parseObservations(`<think>hmm</think> вот: [{"observation":"a","confidence":0.9,"suggested_action":"note"}] всё`)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != 1 || obs[0].Text != "a" {
|
||||||
|
t.Fatalf("got %+v, want one observation 'a'", obs)
|
||||||
|
}
|
||||||
|
|
||||||
|
var b strings.Builder
|
||||||
|
b.WriteString("[")
|
||||||
|
for i := 0; i < MaxObservations+3; i++ {
|
||||||
|
if i > 0 {
|
||||||
|
b.WriteString(",")
|
||||||
|
}
|
||||||
|
b.WriteString(`{"observation":"x","confidence":0.5,"suggested_action":"note"}`)
|
||||||
|
}
|
||||||
|
b.WriteString("]")
|
||||||
|
obs, err = parseObservations(b.String())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse: %v", err)
|
||||||
|
}
|
||||||
|
if len(obs) != MaxObservations {
|
||||||
|
t.Fatalf("parsed %d observations, want the %d cap", len(obs), MaxObservations)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,38 @@
|
|||||||
|
package memory
|
||||||
|
|
||||||
|
// Confidence gate for a recall. Two checks, both must pass before Maven says a
|
||||||
|
// note back:
|
||||||
|
//
|
||||||
|
// - minScore — an absolute cosine floor.
|
||||||
|
// - minMargin — the top hit must beat the runner-up by more than this.
|
||||||
|
//
|
||||||
|
// The margin is the one that carries the weight. The e5 embedder packs every
|
||||||
|
// score into a narrow high band (0.79-0.89 on the recall fixture), so an
|
||||||
|
// absolute floor cannot tell a real hit from a confident-looking miss: every
|
||||||
|
// value under the band admits everything, every value above it answers nothing.
|
||||||
|
// A margin asks a different question — "is this note clearly the best one, or
|
||||||
|
// is the whole shelf equally close?" — and a made-up question has no clear best.
|
||||||
|
//
|
||||||
|
// With one hit and no runner-up there is nothing to compare, so only the floor
|
||||||
|
// applies.
|
||||||
|
|
||||||
|
// ConfidentScores reports whether the top score clears both gates. scores must
|
||||||
|
// be sorted highest first. minMargin <= 0 turns the margin check off.
|
||||||
|
func ConfidentScores(scores []float64, minScore, minMargin float64) bool {
|
||||||
|
if len(scores) == 0 || scores[0] < minScore {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
if minMargin > 0 && len(scores) > 1 && scores[0]-scores[1] <= minMargin {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
|
||||||
|
// Confident is ConfidentScores for search results.
|
||||||
|
func Confident(results []Result, minScore, minMargin float64) bool {
|
||||||
|
scores := make([]float64, len(results))
|
||||||
|
for i, r := range results {
|
||||||
|
scores[i] = r.Score
|
||||||
|
}
|
||||||
|
return ConfidentScores(scores, minScore, minMargin)
|
||||||
|
}
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
package memory
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
func TestConfidentScores(t *testing.T) {
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
scores []float64
|
||||||
|
minScore float64
|
||||||
|
minMargin float64
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{"no hits", nil, 0.55, 0.008, false},
|
||||||
|
{"below the floor", []float64{0.40, 0.10}, 0.55, 0.008, false},
|
||||||
|
{"clear winner", []float64{0.86, 0.70}, 0.55, 0.008, true},
|
||||||
|
{"runner-up too close", []float64{0.860, 0.858}, 0.55, 0.008, false},
|
||||||
|
// The rule is "beats the runner-up by MORE than delta". Not testing an
|
||||||
|
// exactly-equal margin: no pair of these decimals subtracts to exactly
|
||||||
|
// 0.008 in binary float, so such a test would pin rounding, not the rule.
|
||||||
|
{"margin just under delta", []float64{0.8079, 0.8}, 0.55, 0.008, false},
|
||||||
|
{"margin just over delta", []float64{0.8081, 0.8}, 0.55, 0.008, true},
|
||||||
|
// One hit: nothing to compare against, so only the floor applies.
|
||||||
|
{"single hit clears", []float64{0.86}, 0.55, 0.008, true},
|
||||||
|
{"single hit below floor", []float64{0.10}, 0.55, 0.008, false},
|
||||||
|
// Margin off — the old absolute-only behaviour.
|
||||||
|
{"margin off admits a tie", []float64{0.86, 0.86}, 0.55, 0, true},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
if got := ConfidentScores(c.scores, c.minScore, c.minMargin); got != c.want {
|
||||||
|
t.Errorf("got %v, want %v", got, c.want)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestConfidentReadsResultScores(t *testing.T) {
|
||||||
|
res := []Result{{ID: "a", Score: 0.86}, {ID: "b", Score: 0.858}}
|
||||||
|
if Confident(res, 0.55, 0.008) {
|
||||||
|
t.Error("thin margin passed the gate")
|
||||||
|
}
|
||||||
|
if !Confident(res, 0.55, 0) {
|
||||||
|
t.Error("margin off should fall back to the floor alone")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -117,18 +117,40 @@ type cachingEmbedder struct {
|
|||||||
seen map[string][]float32
|
seen map[string][]float32
|
||||||
}
|
}
|
||||||
|
|
||||||
|
var _ router.AsymmetricEmbedder = (*cachingEmbedder)(nil)
|
||||||
|
|
||||||
func (c *cachingEmbedder) Dim() int { return c.inner.Dim() }
|
func (c *cachingEmbedder) Dim() int { return c.inner.Dim() }
|
||||||
func (c *cachingEmbedder) Close() error { return nil } // the caller owns inner
|
func (c *cachingEmbedder) Close() error { return nil } // the caller owns inner
|
||||||
|
|
||||||
func (c *cachingEmbedder) Embed(ctx context.Context, text string) ([]float32, error) {
|
func (c *cachingEmbedder) Embed(ctx context.Context, text string) ([]float32, error) {
|
||||||
if v, ok := c.seen[text]; ok {
|
return c.cached(ctx, "embed:"+text, func() ([]float32, error) {
|
||||||
|
return c.inner.Embed(ctx, text)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// The two sides of an asymmetric embedder give different vectors for the same
|
||||||
|
// string, so the cache key has to say which side asked.
|
||||||
|
func (c *cachingEmbedder) EmbedQuery(ctx context.Context, text string) ([]float32, error) {
|
||||||
|
return c.cached(ctx, "query:"+text, func() ([]float32, error) {
|
||||||
|
return router.EmbedQuery(ctx, c.inner, text)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *cachingEmbedder) EmbedPassage(ctx context.Context, text string) ([]float32, error) {
|
||||||
|
return c.cached(ctx, "passage:"+text, func() ([]float32, error) {
|
||||||
|
return router.EmbedPassage(ctx, c.inner, text)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *cachingEmbedder) cached(_ context.Context, key string, embed func() ([]float32, error)) ([]float32, error) {
|
||||||
|
if v, ok := c.seen[key]; ok {
|
||||||
return v, nil
|
return v, nil
|
||||||
}
|
}
|
||||||
v, err := c.inner.Embed(ctx, text)
|
v, err := embed()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
c.seen[text] = v
|
c.seen[key] = v
|
||||||
return v, nil
|
return v, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -155,6 +177,8 @@ type Outcome struct {
|
|||||||
Tied bool
|
Tied bool
|
||||||
TopID string
|
TopID string
|
||||||
TopScor float64
|
TopScor float64
|
||||||
|
// Margin — top1 − top2. 0 when fewer than two hits came back.
|
||||||
|
Margin float64
|
||||||
Reasons []string
|
Reasons []string
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -162,8 +186,10 @@ type Outcome struct {
|
|||||||
// ranks first but is silenced by query_min_score is a threshold problem, and a
|
// ranks first but is silenced by query_min_score is a threshold problem, and a
|
||||||
// note that never ranks first is an embedder problem. Those are different fixes.
|
// note that never ranks first is an embedder problem. Those are different fixes.
|
||||||
type Report struct {
|
type Report struct {
|
||||||
Name string
|
Name string
|
||||||
MinScore float64
|
MinScore float64
|
||||||
|
// MinMargin — how far the top hit must beat the runner-up. 0 ⇒ off.
|
||||||
|
MinMargin float64
|
||||||
Total int
|
Total int
|
||||||
Answerable int
|
Answerable int
|
||||||
Rank1 int
|
Rank1 int
|
||||||
@@ -190,9 +216,14 @@ type Report struct {
|
|||||||
// where the right note ranked first, and for the no-answer cases. The gap
|
// where the right note ranked first, and for the no-answer cases. The gap
|
||||||
// between these two distributions is what a defensible query_min_score
|
// between these two distributions is what a defensible query_min_score
|
||||||
// would have to sit inside; if they overlap, no threshold separates them.
|
// would have to sit inside; if they overlap, no threshold separates them.
|
||||||
CorrectTop []float64
|
CorrectTop []float64
|
||||||
NoAnswerTop []float64
|
NoAnswerTop []float64
|
||||||
P50, P95, Max time.Duration
|
// CorrectMargin / NoAnswerMargin — the same two groups, but top1 − top2
|
||||||
|
// instead of top1. This is the pair the margin gate has to separate, and
|
||||||
|
// unlike the absolute scores it is what the sweep reads.
|
||||||
|
CorrectMargin []float64
|
||||||
|
NoAnswerMargin []float64
|
||||||
|
P50, P95, Max time.Duration
|
||||||
}
|
}
|
||||||
|
|
||||||
// TagStat — passed/total for one slice of the fixture.
|
// TagStat — passed/total for one slice of the fixture.
|
||||||
@@ -223,13 +254,14 @@ func ratio(n, d int) float64 {
|
|||||||
// the run on an embed or search error: an erroring case scores as a miss and is
|
// the run on an embed or search error: an erroring case scores as a miss and is
|
||||||
// counted in Errors, because "the embedder was down" and "the embedder was
|
// counted in Errors, because "the embedder was down" and "the embedder was
|
||||||
// wrong" are different numbers.
|
// wrong" are different numbers.
|
||||||
func Score(ctx context.Context, name string, emb router.Embedder, newStore NewStore, minScore float64, f Fixture) (Report, error) {
|
func Score(ctx context.Context, name string, emb router.Embedder, newStore NewStore, minScore, minMargin float64, f Fixture) (Report, error) {
|
||||||
rep := Report{
|
rep := Report{
|
||||||
Name: name,
|
Name: name,
|
||||||
MinScore: minScore,
|
MinScore: minScore,
|
||||||
Total: len(f.Cases),
|
MinMargin: minMargin,
|
||||||
ByTag: map[string]TagStat{},
|
Total: len(f.Cases),
|
||||||
ByLang: map[string]TagStat{},
|
ByTag: map[string]TagStat{},
|
||||||
|
ByLang: map[string]TagStat{},
|
||||||
}
|
}
|
||||||
lat := make([]time.Duration, 0, len(f.Cases))
|
lat := make([]time.Duration, 0, len(f.Cases))
|
||||||
|
|
||||||
@@ -239,7 +271,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
|
|||||||
} else {
|
} else {
|
||||||
rep.NoAnswer++
|
rep.NoAnswer++
|
||||||
}
|
}
|
||||||
o, err := scoreCase(ctx, emb, newStore, minScore, c, f.Filler)
|
o, err := scoreCase(ctx, emb, newStore, minScore, minMargin, c, f.Filler)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Report{}, err
|
return Report{}, err
|
||||||
}
|
}
|
||||||
@@ -263,6 +295,11 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
|
|||||||
} else if !o.Rank1 {
|
} else if !o.Rank1 {
|
||||||
rep.WrongTop++
|
rep.WrongTop++
|
||||||
}
|
}
|
||||||
|
if o.Rank1 {
|
||||||
|
// Margins are collected on rank, not on the gate, so the
|
||||||
|
// distribution does not move as the sweep changes the gate.
|
||||||
|
rep.CorrectMargin = append(rep.CorrectMargin, o.Margin)
|
||||||
|
}
|
||||||
if o.Rank1 && o.Recalled != "" {
|
if o.Rank1 && o.Recalled != "" {
|
||||||
rep.CorrectTop = append(rep.CorrectTop, o.TopScor)
|
rep.CorrectTop = append(rep.CorrectTop, o.TopScor)
|
||||||
}
|
}
|
||||||
@@ -271,6 +308,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
|
|||||||
rep.FalseRecall++
|
rep.FalseRecall++
|
||||||
}
|
}
|
||||||
rep.NoAnswerTop = append(rep.NoAnswerTop, o.TopScor)
|
rep.NoAnswerTop = append(rep.NoAnswerTop, o.TopScor)
|
||||||
|
rep.NoAnswerMargin = append(rep.NoAnswerMargin, o.Margin)
|
||||||
}
|
}
|
||||||
|
|
||||||
if o.Pass {
|
if o.Pass {
|
||||||
@@ -285,6 +323,8 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
|
|||||||
|
|
||||||
sort.Float64s(rep.CorrectTop)
|
sort.Float64s(rep.CorrectTop)
|
||||||
sort.Float64s(rep.NoAnswerTop)
|
sort.Float64s(rep.NoAnswerTop)
|
||||||
|
sort.Float64s(rep.CorrectMargin)
|
||||||
|
sort.Float64s(rep.NoAnswerMargin)
|
||||||
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
|
sort.Slice(lat, func(i, j int) bool { return lat[i] < lat[j] })
|
||||||
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
|
rep.P50, rep.P95 = percentile(lat, 0.50), percentile(lat, 0.95)
|
||||||
if len(lat) > 0 {
|
if len(lat) > 0 {
|
||||||
@@ -296,7 +336,7 @@ func Score(ctx context.Context, name string, emb router.Embedder, newStore NewSt
|
|||||||
// scoreCase inserts the case's notes into a fresh store, then runs the read
|
// scoreCase inserts the case's notes into a fresh store, then runs the read
|
||||||
// path the daemon runs. The returned error is fatal (the harness is broken);
|
// path the daemon runs. The returned error is fatal (the harness is broken);
|
||||||
// an embedder or store failure on the query lands in Outcome.Err instead.
|
// an embedder or store failure on the query lands in Outcome.Err instead.
|
||||||
func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minScore float64, c Case, filler []StoredNote) (Outcome, error) {
|
func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minScore, minMargin float64, c Case, filler []StoredNote) (Outcome, error) {
|
||||||
st, release, err := newStore()
|
st, release, err := newStore()
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Outcome{}, fmt.Errorf("%s: new store: %w", c.ID, err)
|
return Outcome{}, fmt.Errorf("%s: new store: %w", c.ID, err)
|
||||||
@@ -305,7 +345,7 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
|
|||||||
|
|
||||||
all := append(append([]StoredNote(nil), c.Notes...), filler...)
|
all := append(append([]StoredNote(nil), c.Notes...), filler...)
|
||||||
for _, n := range all {
|
for _, n := range all {
|
||||||
vec, err := emb.Embed(ctx, n.Text)
|
vec, err := router.EmbedPassage(ctx, emb, n.Text)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Outcome{}, fmt.Errorf("%s: embed note %s: %w", c.ID, n.ID, err)
|
return Outcome{}, fmt.Errorf("%s: embed note %s: %w", c.ID, n.ID, err)
|
||||||
}
|
}
|
||||||
@@ -317,7 +357,7 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
|
|||||||
|
|
||||||
o := Outcome{Case: c}
|
o := Outcome{Case: c}
|
||||||
start := time.Now()
|
start := time.Now()
|
||||||
qvec, err := emb.Embed(ctx, c.Query)
|
qvec, err := router.EmbedQuery(ctx, emb, c.Query)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
o.Latency = time.Since(start)
|
o.Latency = time.Since(start)
|
||||||
o.Err = err
|
o.Err = err
|
||||||
@@ -335,7 +375,10 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
|
|||||||
|
|
||||||
if len(hits) > 0 {
|
if len(hits) > 0 {
|
||||||
o.TopID, o.TopScor = hits[0].ID, hits[0].Score
|
o.TopID, o.TopScor = hits[0].ID, hits[0].Score
|
||||||
o.Recalled = bestRecall(hits, minScore)
|
if len(hits) > 1 {
|
||||||
|
o.Margin = hits[0].Score - hits[1].Score
|
||||||
|
}
|
||||||
|
o.Recalled = bestRecall(hits, minScore, minMargin)
|
||||||
}
|
}
|
||||||
for i, h := range hits {
|
for i, h := range hits {
|
||||||
if h.ID != c.Want {
|
if h.ID != c.Want {
|
||||||
@@ -356,14 +399,14 @@ func scoreCase(ctx context.Context, emb router.Embedder, newStore NewStore, minS
|
|||||||
switch {
|
switch {
|
||||||
case !c.Answerable():
|
case !c.Answerable():
|
||||||
if o.Recalled != "" {
|
if o.Recalled != "" {
|
||||||
o.Reasons = append(o.Reasons, fmt.Sprintf("false recall: %q at %.3f, want silence", o.TopID, o.TopScor))
|
o.Reasons = append(o.Reasons, fmt.Sprintf("false recall: %q at %.3f (margin %.3f), want silence", o.TopID, o.TopScor, o.Margin))
|
||||||
}
|
}
|
||||||
case o.Tied:
|
case o.Tied:
|
||||||
o.Reasons = append(o.Reasons, fmt.Sprintf("tie at %.3f — the right note is on top only by sort order", o.TopScor))
|
o.Reasons = append(o.Reasons, fmt.Sprintf("tie at %.3f — the right note is on top only by sort order", o.TopScor))
|
||||||
case !o.Rank1:
|
case !o.Rank1:
|
||||||
o.Reasons = append(o.Reasons, fmt.Sprintf("top hit %q (%.3f), want %q%s", o.TopID, o.TopScor, c.Want, rankNote(o.Rank3)))
|
o.Reasons = append(o.Reasons, fmt.Sprintf("top hit %q (%.3f), want %q%s", o.TopID, o.TopScor, c.Want, rankNote(o.Rank3)))
|
||||||
case o.Recalled == "":
|
case o.Recalled == "":
|
||||||
o.Reasons = append(o.Reasons, fmt.Sprintf("right note ranked first but scored %.3f < gate %.2f — daemon says \"не знаю\"", o.TopScor, minScore))
|
o.Reasons = append(o.Reasons, fmt.Sprintf("right note ranked first at %.3f (margin %.3f) but the gate silenced it — daemon says \"не знаю\"", o.TopScor, o.Margin))
|
||||||
}
|
}
|
||||||
o.Pass = len(o.Reasons) == 0
|
o.Pass = len(o.Reasons) == 0
|
||||||
return o, nil
|
return o, nil
|
||||||
@@ -379,8 +422,10 @@ func rankNote(inTop3 bool) string {
|
|||||||
// bestRecall mirrors cmd/mavend/recall.go — the gate the daemon actually
|
// bestRecall mirrors cmd/mavend/recall.go — the gate the daemon actually
|
||||||
// applies to a memory hit. Duplicated rather than imported because package main
|
// applies to a memory hit. Duplicated rather than imported because package main
|
||||||
// is not importable; recalleval_test.go asserts the two agree in behaviour.
|
// is not importable; recalleval_test.go asserts the two agree in behaviour.
|
||||||
func bestRecall(results []memory.Result, min float64) string {
|
// The daemon returns the whole hit (a note and a fact are said differently);
|
||||||
if len(results) == 0 || results[0].Score < min {
|
// the harness only scores what came back, so it keeps returning the text.
|
||||||
|
func bestRecall(results []memory.Result, minScore, minMargin float64) string {
|
||||||
|
if !memory.Confident(results, minScore, minMargin) {
|
||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
return results[0].Meta["text"]
|
return results[0].Meta["text"]
|
||||||
@@ -415,7 +460,7 @@ func percentile(sorted []time.Duration, p float64) time.Duration {
|
|||||||
// the slices that name where the path is weak.
|
// the slices that name where the path is weak.
|
||||||
func (r Report) String() string {
|
func (r Report) String() string {
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
fmt.Fprintf(&b, "%s: %d/%d cases pass (gate %.2f)\n", r.Name, r.Passed, r.Total, r.MinScore)
|
fmt.Fprintf(&b, "%s: %d/%d cases pass (gate %.2f, margin %.3f)\n", r.Name, r.Passed, r.Total, r.MinScore, r.MinMargin)
|
||||||
fmt.Fprintf(&b, " recall@1 %.1f%% (%d/%d) recall@3 %.1f%% (%d/%d) answered after gate %.1f%% (%d/%d)\n",
|
fmt.Fprintf(&b, " recall@1 %.1f%% (%d/%d) recall@3 %.1f%% (%d/%d) answered after gate %.1f%% (%d/%d)\n",
|
||||||
100*r.Recall1(), r.Rank1, r.Answerable,
|
100*r.Recall1(), r.Rank1, r.Answerable,
|
||||||
100*r.Recall3(), r.Rank3, r.Answerable,
|
100*r.Recall3(), r.Rank3, r.Answerable,
|
||||||
@@ -426,6 +471,8 @@ func (r Report) String() string {
|
|||||||
100*r.FalseRecallRate(), r.FalseRecall, r.NoAnswer)
|
100*r.FalseRecallRate(), r.FalseRecall, r.NoAnswer)
|
||||||
fmt.Fprintf(&b, " top-1 score, right note first: %s\n", spread(r.CorrectTop))
|
fmt.Fprintf(&b, " top-1 score, right note first: %s\n", spread(r.CorrectTop))
|
||||||
fmt.Fprintf(&b, " top-1 score, must be silent: %s\n", spread(r.NoAnswerTop))
|
fmt.Fprintf(&b, " top-1 score, must be silent: %s\n", spread(r.NoAnswerTop))
|
||||||
|
fmt.Fprintf(&b, " margin top1-top2, right note first: %s\n", spread(r.CorrectMargin))
|
||||||
|
fmt.Fprintf(&b, " margin top1-top2, must be silent: %s\n", spread(r.NoAnswerMargin))
|
||||||
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
|
fmt.Fprintf(&b, " latency: p50 %s p95 %s max %s\n", r.P50, r.P95, r.Max)
|
||||||
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
|
fmt.Fprintf(&b, " by lang: %s\n", renderStats(r.ByLang))
|
||||||
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
|
fmt.Fprintf(&b, " by tag: %s\n", renderStats(r.ByTag))
|
||||||
|
|||||||
@@ -142,21 +142,40 @@ func words(s string) []string {
|
|||||||
// cmd/mavend/recall.go (package main is not importable). This pins the copy to
|
// cmd/mavend/recall.go (package main is not importable). This pins the copy to
|
||||||
// the original's three rules: no hits, below the gate, or no text ⇒ silence.
|
// the original's three rules: no hits, below the gate, or no text ⇒ silence.
|
||||||
func TestBestRecallMatchesDaemon(t *testing.T) {
|
func TestBestRecallMatchesDaemon(t *testing.T) {
|
||||||
if got := bestRecall(nil, 0.55); got != "" {
|
if got := bestRecall(nil, 0.55, 0); got != "" {
|
||||||
t.Errorf("no hits: got %q, want silence", got)
|
t.Errorf("no hits: got %q, want silence", got)
|
||||||
}
|
}
|
||||||
low := []memory.Result{{ID: "a", Score: 0.4, Meta: map[string]string{"text": "чай"}}}
|
low := []memory.Result{{ID: "a", Score: 0.4, Meta: map[string]string{"text": "чай"}}}
|
||||||
if got := bestRecall(low, 0.55); got != "" {
|
if got := bestRecall(low, 0.55, 0); got != "" {
|
||||||
t.Errorf("below gate: got %q, want silence", got)
|
t.Errorf("below gate: got %q, want silence", got)
|
||||||
}
|
}
|
||||||
noText := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{}}}
|
noText := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{}}}
|
||||||
if got := bestRecall(noText, 0.55); got != "" {
|
if got := bestRecall(noText, 0.55, 0); got != "" {
|
||||||
t.Errorf("no text: got %q, want silence", got)
|
t.Errorf("no text: got %q, want silence", got)
|
||||||
}
|
}
|
||||||
ok := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{"text": "чай"}}}
|
ok := []memory.Result{{ID: "a", Score: 0.9, Meta: map[string]string{"text": "чай"}}}
|
||||||
if got := bestRecall(ok, 0.55); got != "чай" {
|
if got := bestRecall(ok, 0.55, 0); got != "чай" {
|
||||||
t.Errorf("above gate: got %q, want %q", got, "чай")
|
t.Errorf("above gate: got %q, want %q", got, "чай")
|
||||||
}
|
}
|
||||||
|
// Margin: a close runner-up means the embedder cannot tell the two apart,
|
||||||
|
// so Maven stays silent even though both clear the absolute floor.
|
||||||
|
close := []memory.Result{
|
||||||
|
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
|
||||||
|
{ID: "b", Score: 0.85, Meta: map[string]string{"text": "кофе"}},
|
||||||
|
}
|
||||||
|
if got := bestRecall(close, 0.55, 0.03); got != "" {
|
||||||
|
t.Errorf("thin margin: got %q, want silence", got)
|
||||||
|
}
|
||||||
|
if got := bestRecall(close, 0.55, 0); got != "чай" {
|
||||||
|
t.Errorf("margin off: got %q, want %q", got, "чай")
|
||||||
|
}
|
||||||
|
clear := []memory.Result{
|
||||||
|
{ID: "a", Score: 0.86, Meta: map[string]string{"text": "чай"}},
|
||||||
|
{ID: "b", Score: 0.70, Meta: map[string]string{"text": "кофе"}},
|
||||||
|
}
|
||||||
|
if got := bestRecall(clear, 0.55, 0.03); got != "чай" {
|
||||||
|
t.Errorf("wide margin: got %q, want %q", got, "чай")
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestHashRecallBaseline — the CI ratchet. HashEmbedder, so it needs no model
|
// TestHashRecallBaseline — the CI ratchet. HashEmbedder, so it needs no model
|
||||||
@@ -171,14 +190,15 @@ func TestHashRecallBaseline(t *testing.T) {
|
|||||||
t.Fatalf("Load: %v", err)
|
t.Fatalf("Load: %v", err)
|
||||||
}
|
}
|
||||||
rep, err := Score(context.Background(), "recall+hash", router.NewHashEmbedder(hashDim), InMemory,
|
rep, err := Score(context.Background(), "recall+hash", router.NewHashEmbedder(hashDim), InMemory,
|
||||||
config.DefaultQueryMinScore, f)
|
config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score: %v", err)
|
t.Fatalf("Score: %v", err)
|
||||||
}
|
}
|
||||||
t.Log("\n" + rep.String() + rep.Failures())
|
t.Log("\n" + rep.String() + rep.Failures())
|
||||||
t.Log("\ngate sweep:\n" + sweep(t, router.NewHashEmbedder(hashDim), f))
|
t.Log("\ngate sweep:\n" + sweep(t, router.NewHashEmbedder(hashDim), f))
|
||||||
|
|
||||||
// 0.32 sits under the observed 0.360 recall@1.
|
// 0.32 sits under the observed 0.370 recall@1 (was 0.360 over 25 answerable
|
||||||
|
// cases; the two mixed note+fact cases added with #373 make it 27).
|
||||||
const floorRecall1 = 0.32
|
const floorRecall1 = 0.32
|
||||||
if rep.Recall1() < floorRecall1 {
|
if rep.Recall1() < floorRecall1 {
|
||||||
t.Errorf("recall@1 %.3f below ratchet %.2f — note recall regressed", rep.Recall1(), floorRecall1)
|
t.Errorf("recall@1 %.3f below ratchet %.2f — note recall regressed", rep.Recall1(), floorRecall1)
|
||||||
@@ -201,11 +221,11 @@ func TestPersistentStoreScoresTheSame(t *testing.T) {
|
|||||||
t.Fatalf("Load: %v", err)
|
t.Fatalf("Load: %v", err)
|
||||||
}
|
}
|
||||||
emb := router.NewHashEmbedder(hashDim)
|
emb := router.NewHashEmbedder(hashDim)
|
||||||
inMem, err := Score(context.Background(), "recall+hash+memory", emb, InMemory, config.DefaultQueryMinScore, f)
|
inMem, err := Score(context.Background(), "recall+hash+memory", emb, InMemory, config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score in-memory: %v", err)
|
t.Fatalf("Score in-memory: %v", err)
|
||||||
}
|
}
|
||||||
persistent, err := Score(context.Background(), "recall+hash+sqlite", emb, sqliteStores(t), config.DefaultQueryMinScore, f)
|
persistent, err := Score(context.Background(), "recall+hash+sqlite", emb, sqliteStores(t), config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score sqlite: %v", err)
|
t.Fatalf("Score sqlite: %v", err)
|
||||||
}
|
}
|
||||||
@@ -246,8 +266,8 @@ func TestONNXRecall(t *testing.T) {
|
|||||||
if lib == "" {
|
if lib == "" {
|
||||||
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
t.Skip("MAVEN_ONNX_LIB unset — see AGENTS.md § Embedder model for intent routing")
|
||||||
}
|
}
|
||||||
model := filepath.Join("../../..", "models/embedder/model.onnx")
|
model := filepath.Join("../../..", "models/embedder/multilingual-e5-small/model_quantized.onnx")
|
||||||
tok := filepath.Join("../../..", "models/embedder/tokenizer.json")
|
tok := filepath.Join("../../..", "models/embedder/multilingual-e5-small/tokenizer.json")
|
||||||
for _, p := range []string{lib, model, tok} {
|
for _, p := range []string{lib, model, tok} {
|
||||||
if _, err := os.Stat(p); err != nil {
|
if _, err := os.Stat(p); err != nil {
|
||||||
t.Skipf("missing %s: %v", p, err)
|
t.Skipf("missing %s: %v", p, err)
|
||||||
@@ -263,14 +283,16 @@ func TestONNXRecall(t *testing.T) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Load: %v", err)
|
t.Fatalf("Load: %v", err)
|
||||||
}
|
}
|
||||||
rep, err := Score(context.Background(), "recall+onnx", emb, InMemory, config.DefaultQueryMinScore, f)
|
rep, err := Score(context.Background(), "recall+onnx", emb, InMemory, config.DefaultQueryMinScore, config.DefaultQueryMinMargin, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Score: %v", err)
|
t.Fatalf("Score: %v", err)
|
||||||
}
|
}
|
||||||
t.Log("\n" + rep.String() + rep.Failures())
|
t.Log("\n" + rep.String() + rep.Failures())
|
||||||
// Cached for the sweep only: the headline run above must pay the real
|
// Cached for the sweeps only: the headline run above must pay the real
|
||||||
// embedder cost so its latency numbers mean something.
|
// embedder cost so its latency numbers mean something.
|
||||||
t.Log("\ngate sweep:\n" + sweep(t, Cache(emb), f))
|
cached := Cache(emb)
|
||||||
|
t.Log("\ngate sweep (margin off):\n" + sweep(t, cached, f))
|
||||||
|
t.Log("\nmargin sweep (gate 0.55):\n" + marginSweep(t, cached, f))
|
||||||
}
|
}
|
||||||
|
|
||||||
// sweep scores the fixture at a range of gates and renders one line each. Two
|
// sweep scores the fixture at a range of gates and renders one line each. Two
|
||||||
@@ -281,7 +303,7 @@ func sweep(t *testing.T, emb router.Embedder, f Fixture) string {
|
|||||||
t.Helper()
|
t.Helper()
|
||||||
var b strings.Builder
|
var b strings.Builder
|
||||||
for _, gate := range []float64{0.0, 0.30, 0.40, 0.50, 0.55, 0.60, 0.70, 0.80, 0.90} {
|
for _, gate := range []float64{0.0, 0.30, 0.40, 0.50, 0.55, 0.60, 0.70, 0.80, 0.90} {
|
||||||
rep, err := Score(context.Background(), "sweep", emb, InMemory, gate, f)
|
rep, err := Score(context.Background(), "sweep", emb, InMemory, gate, 0, f)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("sweep at %.2f: %v", gate, err)
|
t.Fatalf("sweep at %.2f: %v", gate, err)
|
||||||
}
|
}
|
||||||
@@ -290,3 +312,21 @@ func sweep(t *testing.T, emb router.Embedder, f Fixture) string {
|
|||||||
}
|
}
|
||||||
return b.String()
|
return b.String()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// marginSweep is the same idea for the margin gate (top1 − top2 > delta), with
|
||||||
|
// the absolute gate held at its default. The absolute score cannot separate a
|
||||||
|
// real hit from a made-up question under e5 — every score lands in one narrow
|
||||||
|
// band — so this sweep is the one that picks a number.
|
||||||
|
func marginSweep(t *testing.T, emb router.Embedder, f Fixture) string {
|
||||||
|
t.Helper()
|
||||||
|
var b strings.Builder
|
||||||
|
for _, d := range []float64{0, 0.002, 0.005, 0.008, 0.01, 0.012, 0.015, 0.02, 0.025, 0.03, 0.04, 0.05, 0.06} {
|
||||||
|
rep, err := Score(context.Background(), "margin sweep", emb, InMemory, config.DefaultQueryMinScore, d, f)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("margin sweep at %.3f: %v", d, err)
|
||||||
|
}
|
||||||
|
fmt.Fprintf(&b, " delta %.3f: answered %d/%d (%.0f%%) false recall %d/%d\n",
|
||||||
|
d, rep.Rank1-rep.Gated, rep.Answerable, 100*rep.Answered(), rep.FalseRecall, rep.NoAnswer)
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|||||||
@@ -387,6 +387,32 @@
|
|||||||
{"id": "n2", "text": "wifi channel is 6", "kind": "note"},
|
{"id": "n2", "text": "wifi channel is 6", "kind": "note"},
|
||||||
{"id": "n3", "text": "the guest network is off", "kind": "note"}
|
{"id": "n3", "text": "the guest network is off", "kind": "note"}
|
||||||
]
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "ru-mixed-031",
|
||||||
|
"lang": "ru",
|
||||||
|
"tags": ["mixed", "paraphrase", "hard"],
|
||||||
|
"note": "notes and facts in one store and the note is the answer — the daemon indexes both (Vikunja #373)",
|
||||||
|
"query": "куда я спрятал второй ключ от квартиры",
|
||||||
|
"want": "n1",
|
||||||
|
"notes": [
|
||||||
|
{"id": "n1", "text": "запасной ключ от квартиры лежит в синей коробке на полке", "kind": "note"},
|
||||||
|
{"id": "x1", "text": "поменял замок в двери двадцатого июня", "kind": "fact"},
|
||||||
|
{"id": "x2", "text": "отдал ключ соседке в мае", "kind": "fact"}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"id": "ru-mixed-032",
|
||||||
|
"lang": "ru",
|
||||||
|
"tags": ["mixed", "distractor"],
|
||||||
|
"note": "the mirror of ru-mixed-031: the fact answers and the notes are the distractors",
|
||||||
|
"query": "когда я в последний раз заливал бензин",
|
||||||
|
"want": "x1",
|
||||||
|
"notes": [
|
||||||
|
{"id": "x1", "text": "залил полный бак в четверг вечером", "kind": "fact"},
|
||||||
|
{"id": "n1", "text": "на заправке у моста дешевле бензин", "kind": "note"},
|
||||||
|
{"id": "n2", "text": "надо поменять зимние шины", "kind": "note"}
|
||||||
|
]
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -20,9 +20,19 @@ type ProposedRoutine struct {
|
|||||||
const MaxIntervalRatio = 1.5
|
const MaxIntervalRatio = 1.5
|
||||||
|
|
||||||
// MinEvents is the minimum number of events needed to detect a pattern.
|
// MinEvents is the minimum number of events needed to detect a pattern.
|
||||||
// With N events, there are N-1 intervals; we need at least 2 intervals
|
// With N events there are N-1 intervals, so 4 events means 3 intervals.
|
||||||
// before proposing anything.
|
//
|
||||||
const MinEvents = 3
|
// This used to be 3 (two intervals), which is not a pattern — it is a
|
||||||
|
// coincidence with a mean. Two gaps of similar length happen constantly:
|
||||||
|
// water the plants on a Sunday, again the next Sunday, once more the Sunday
|
||||||
|
// after, and a detector with a ±50% band calls that a weekly routine. The
|
||||||
|
// cost of being wrong is asymmetric now that the digestion tick scans all of
|
||||||
|
// history on its own schedule and can announce what it finds: a false
|
||||||
|
// positive is something the owner has to read and dismiss, and a dismissal
|
||||||
|
// is permanent, so one bad guess burns that action+object pair forever.
|
||||||
|
// Three intervals is the cheapest bar that makes a run distinguishable from
|
||||||
|
// a repeat. False negatives cost one more observation and nothing else.
|
||||||
|
const MinEvents = 4
|
||||||
|
|
||||||
// Detect checks whether a sequence of events for the same action+object
|
// Detect checks whether a sequence of events for the same action+object
|
||||||
// forms a stable recurring pattern. Returns a ProposedRoutine when:
|
// forms a stable recurring pattern. Returns a ProposedRoutine when:
|
||||||
|
|||||||
@@ -6,12 +6,13 @@ import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
func TestDetectEnoughEvents(t *testing.T) {
|
func TestDetectEnoughEvents(t *testing.T) {
|
||||||
// 3 events with 7-day intervals → stable pattern
|
// MinEvents events with 7-day intervals → stable pattern
|
||||||
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
||||||
events := []Event{
|
events := []Event{
|
||||||
{Action: "refill", Object: "cat_water", Ts: base},
|
{Action: "refill", Object: "cat_water", Ts: base},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(7 * 24 * time.Hour)},
|
{Action: "refill", Object: "cat_water", Ts: base.Add(7 * 24 * time.Hour)},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(14 * 24 * time.Hour)},
|
{Action: "refill", Object: "cat_water", Ts: base.Add(14 * 24 * time.Hour)},
|
||||||
|
{Action: "refill", Object: "cat_water", Ts: base.Add(21 * 24 * time.Hour)},
|
||||||
}
|
}
|
||||||
|
|
||||||
r, err := Detect(events)
|
r, err := Detect(events)
|
||||||
@@ -24,8 +25,8 @@ func TestDetectEnoughEvents(t *testing.T) {
|
|||||||
if r.Action != "refill" || r.Object != "cat_water" {
|
if r.Action != "refill" || r.Object != "cat_water" {
|
||||||
t.Fatalf("action/object: want refill/cat_water, got %s/%s", r.Action, r.Object)
|
t.Fatalf("action/object: want refill/cat_water, got %s/%s", r.Action, r.Object)
|
||||||
}
|
}
|
||||||
if r.N != 3 {
|
if r.N != 4 {
|
||||||
t.Fatalf("want N=3, got %d", r.N)
|
t.Fatalf("want N=4, got %d", r.N)
|
||||||
}
|
}
|
||||||
// ~7 days
|
// ~7 days
|
||||||
if r.IntervalDays < 6.9 || r.IntervalDays > 7.1 {
|
if r.IntervalDays < 6.9 || r.IntervalDays > 7.1 {
|
||||||
@@ -33,19 +34,28 @@ func TestDetectEnoughEvents(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestDetectNotEnoughEvents — two intervals are a coincidence, not a routine
|
||||||
|
// (Vikunja #43). Three same-day-of-week events used to be enough to propose a
|
||||||
|
// weekly reminder; MinEvents is 4 now so a repeat has to happen a third time
|
||||||
|
// before Maven calls it a pattern.
|
||||||
func TestDetectNotEnoughEvents(t *testing.T) {
|
func TestDetectNotEnoughEvents(t *testing.T) {
|
||||||
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
||||||
events := []Event{
|
for _, n := range []int{1, 2, MinEvents - 1} {
|
||||||
{Action: "refill", Object: "cat_water", Ts: base},
|
events := make([]Event, n)
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(7 * 24 * time.Hour)},
|
for i := range events {
|
||||||
}
|
events[i] = Event{
|
||||||
|
Action: "refill",
|
||||||
r, err := Detect(events)
|
Object: "cat_water",
|
||||||
if err != nil {
|
Ts: base.Add(time.Duration(i) * 7 * 24 * time.Hour),
|
||||||
t.Fatalf("Detect: %v", err)
|
}
|
||||||
}
|
}
|
||||||
if r != nil {
|
r, err := Detect(events)
|
||||||
t.Fatal("want nil for <3 events")
|
if err != nil {
|
||||||
|
t.Fatalf("Detect(%d events): %v", n, err)
|
||||||
|
}
|
||||||
|
if r != nil {
|
||||||
|
t.Fatalf("Detect(%d events) proposed %+v, want nil below MinEvents=%d", n, r, MinEvents)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -68,12 +78,13 @@ func TestDetectEmpty(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestDetectIrregularRejects(t *testing.T) {
|
func TestDetectIrregularRejects(t *testing.T) {
|
||||||
// 3 events but wildly irregular: 1 day, then 14 days → ratio 14 > 1.5
|
// wildly irregular: 1 day, then 14 days → ratio 14 > 1.5
|
||||||
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
base := time.Date(2026, 7, 1, 12, 0, 0, 0, time.UTC)
|
||||||
events := []Event{
|
events := []Event{
|
||||||
{Action: "refill", Object: "cat_water", Ts: base},
|
{Action: "refill", Object: "cat_water", Ts: base},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(1 * 24 * time.Hour)},
|
{Action: "refill", Object: "cat_water", Ts: base.Add(1 * 24 * time.Hour)},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(15 * 24 * time.Hour)},
|
{Action: "refill", Object: "cat_water", Ts: base.Add(15 * 24 * time.Hour)},
|
||||||
|
{Action: "refill", Object: "cat_water", Ts: base.Add(16 * 24 * time.Hour)},
|
||||||
}
|
}
|
||||||
|
|
||||||
r, err := Detect(events)
|
r, err := Detect(events)
|
||||||
@@ -117,6 +128,7 @@ func TestDetectSameTimestamp(t *testing.T) {
|
|||||||
{Action: "refill", Object: "cat_water", Ts: base},
|
{Action: "refill", Object: "cat_water", Ts: base},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base},
|
{Action: "refill", Object: "cat_water", Ts: base},
|
||||||
{Action: "refill", Object: "cat_water", Ts: base.Add(7 * 24 * time.Hour)},
|
{Action: "refill", Object: "cat_water", Ts: base.Add(7 * 24 * time.Hour)},
|
||||||
|
{Action: "refill", Object: "cat_water", Ts: base.Add(14 * 24 * time.Hour)},
|
||||||
}
|
}
|
||||||
|
|
||||||
r, err := Detect(events)
|
r, err := Detect(events)
|
||||||
|
|||||||
@@ -0,0 +1,138 @@
|
|||||||
|
// Package persona builds the one shared context block that goes in front of
|
||||||
|
// every LLM system prompt: who the owner is, how to address him, and what
|
||||||
|
// time it is right now.
|
||||||
|
//
|
||||||
|
// Why one block and not a line pasted into each prompt: there are five
|
||||||
|
// prompts (nudges, action replies, chat, note queries, general knowledge) and
|
||||||
|
// the "address him as ты" rule had only reached two of them. Five copies drift.
|
||||||
|
// One block cannot.
|
||||||
|
//
|
||||||
|
// The rules here are defaults in code, not config. Maven is feminine and the
|
||||||
|
// owner is a man addressed informally — that is a hard constraint of the
|
||||||
|
// product, so it must hold with an empty config file. Config only ADDS
|
||||||
|
// optional facts (his name, his city).
|
||||||
|
package persona
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Facts — the optional, deployment-specific half of the block. All fields may
|
||||||
|
// be empty; the block is still correct and useful without them.
|
||||||
|
type Facts struct {
|
||||||
|
OwnerName string // his name, e.g. "Ками"
|
||||||
|
City string // where he is, e.g. "Москва"
|
||||||
|
Static string // the free-text `persona` config string, appended verbatim
|
||||||
|
|
||||||
|
// The two config-gated capabilities. They are listed only when this
|
||||||
|
// deployment actually has them, because a capability she names and cannot
|
||||||
|
// do is worse than one she never mentions.
|
||||||
|
Weather bool // an open-meteo provider is configured
|
||||||
|
Telegram bool // a telegram bot token + chat id are configured
|
||||||
|
Tools bool // at least one shell act is on the allowlist
|
||||||
|
}
|
||||||
|
|
||||||
|
var ruWeekdays = [...]string{"воскресенье", "понедельник", "вторник", "среда", "четверг", "пятница", "суббота"}
|
||||||
|
|
||||||
|
var ruMonths = [...]string{
|
||||||
|
"января", "февраля", "марта", "апреля", "мая", "июня",
|
||||||
|
"июля", "августа", "сентября", "октября", "ноября", "декабря",
|
||||||
|
}
|
||||||
|
|
||||||
|
// Block renders the context block for one turn. Russian even in front of the
|
||||||
|
// English prompts: the rules it states are Russian grammar (ты/тебя, feminine
|
||||||
|
// verbs), and a Russian rule reads best stated in Russian.
|
||||||
|
//
|
||||||
|
// Keep it short. It ships on every turn to a 0.8B on laptop CPU, so every
|
||||||
|
// line here is latency.
|
||||||
|
func (f Facts) Block(now time.Time) string {
|
||||||
|
var b strings.Builder
|
||||||
|
|
||||||
|
b.WriteString("Ты — Maven, домашняя ассистентка. О себе говоришь в женском роде: \"я записала\", \"я проверила\".\n")
|
||||||
|
|
||||||
|
// The address form gets its own line. It is the thing that kept getting
|
||||||
|
// lost when it was buried in prose.
|
||||||
|
b.WriteString("ОБРАЩЕНИЕ: владелец — мужчина, всегда на \"ты\" (ты, тебя, тебе, твой) и в единственном числе (\"выпей\", \"посмотри\"). Никогда \"вы\"/\"вас\"/\"ваш\". Никогда \"он\"/\"его\" о нём — ты говоришь ему, а не о нём. Глаголы о нём — в мужском роде (\"ты забыл\").\n")
|
||||||
|
|
||||||
|
if who := f.who(); who != "" {
|
||||||
|
b.WriteString(who + "\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
b.WriteString(fmt.Sprintf("Сейчас: %s, %d %s %d, %02d:%02d (местное время).\n",
|
||||||
|
ruWeekdays[int(now.Weekday())], now.Day(), ruMonths[int(now.Month())-1], now.Year(),
|
||||||
|
now.Hour(), now.Minute()))
|
||||||
|
|
||||||
|
b.WriteString("Умеешь: " + strings.Join(f.can(), "; ") +
|
||||||
|
". Других ДЕЙСТВИЙ не умеешь — если просят такое, скажи прямо.\n")
|
||||||
|
|
||||||
|
if s := strings.TrimSpace(f.Static); s != "" {
|
||||||
|
b.WriteString(s + "\n")
|
||||||
|
}
|
||||||
|
return b.String()
|
||||||
|
}
|
||||||
|
|
||||||
|
// can lists what she can really do. Every entry here is a code path that
|
||||||
|
// exists in the daemon today:
|
||||||
|
// - reminders: IntentReminder → CoreAPI.CreateReminder, fired by the tick.
|
||||||
|
// - notes and facts: IntentNote/IntentFact write, IntentQuery reads them back.
|
||||||
|
// - calendar: IntentQuery answers "что у меня сегодня" from CalendarEvents.
|
||||||
|
// - weather / telegram / shell acts: only when configured (see Facts).
|
||||||
|
//
|
||||||
|
// Nothing speculative goes in this list. A capability she offers and cannot
|
||||||
|
// perform is worse than one she never mentions.
|
||||||
|
func (f Facts) can() []string {
|
||||||
|
c := []string{
|
||||||
|
// Talking comes first, and the closing line says "действий" rather than
|
||||||
|
// "ничего", because this same block sits in front of the chat and
|
||||||
|
// general-knowledge prompts. A flat "you can do nothing else" would
|
||||||
|
// tell her to refuse the exact thing those two prompts are for.
|
||||||
|
"разговаривать и отвечать на вопросы",
|
||||||
|
"ставить напоминания",
|
||||||
|
"записывать заметки и факты и отвечать по ним",
|
||||||
|
"смотреть календарь",
|
||||||
|
}
|
||||||
|
if f.Weather {
|
||||||
|
c = append(c, "говорить погоду")
|
||||||
|
}
|
||||||
|
if f.Telegram {
|
||||||
|
c = append(c, "писать в телеграм")
|
||||||
|
}
|
||||||
|
if f.Tools {
|
||||||
|
c = append(c, "запускать разрешённые команды на сервере")
|
||||||
|
}
|
||||||
|
return c
|
||||||
|
}
|
||||||
|
|
||||||
|
// who renders the optional name/city line, or "" when neither is configured.
|
||||||
|
//
|
||||||
|
// Written as labels ("Имя владельца: ..."), not as a sentence with pronouns:
|
||||||
|
// the block's own "ты" is Maven, so "тебя зовут" would read as her name and
|
||||||
|
// "его" would model the third-person form she must never use about him.
|
||||||
|
func (f Facts) who() string {
|
||||||
|
name := strings.TrimSpace(f.OwnerName)
|
||||||
|
city := strings.TrimSpace(f.City)
|
||||||
|
switch {
|
||||||
|
case name != "" && city != "":
|
||||||
|
return "Имя владельца: " + name + ". Город: " + city + "."
|
||||||
|
case name != "":
|
||||||
|
return "Имя владельца: " + name + "."
|
||||||
|
case city != "":
|
||||||
|
return "Город: " + city + "."
|
||||||
|
}
|
||||||
|
return ""
|
||||||
|
}
|
||||||
|
|
||||||
|
// Prepend puts the block in front of a system prompt. Nil-safe: a nil renderer
|
||||||
|
// (tests, the stub paths) returns the prompt untouched.
|
||||||
|
func Prepend(block func() string, prompt string) string {
|
||||||
|
if block == nil {
|
||||||
|
return prompt
|
||||||
|
}
|
||||||
|
s := strings.TrimSpace(block())
|
||||||
|
if s == "" {
|
||||||
|
return prompt
|
||||||
|
}
|
||||||
|
return s + "\n\n" + prompt
|
||||||
|
}
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
package persona
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
var ref = time.Date(2026, 7, 31, 14, 5, 0, 0, time.UTC)
|
||||||
|
|
||||||
|
// The block must be correct with an empty config: the address form and the
|
||||||
|
// gender rules are hard constraints, not preferences.
|
||||||
|
func TestBlockWorksWithZeroConfig(t *testing.T) {
|
||||||
|
b := Facts{}.Block(ref)
|
||||||
|
for _, want := range []string{"женском роде", "ОБРАЩЕНИЕ", "\"ты\"", "31 июля 2026", "пятница", "14:05"} {
|
||||||
|
if !strings.Contains(b, want) {
|
||||||
|
t.Errorf("block missing %q:\n%s", want, b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestBlockAddsOptionalFacts(t *testing.T) {
|
||||||
|
b := Facts{OwnerName: "Ками", City: "Москва", Static: "Будь краткой."}.Block(ref)
|
||||||
|
for _, want := range []string{"Ками", "Москва", "Будь краткой."} {
|
||||||
|
if !strings.Contains(b, want) {
|
||||||
|
t.Errorf("block missing %q:\n%s", want, b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The time changes between turns, so two renders must differ.
|
||||||
|
func TestBlockRendersTimePerTurn(t *testing.T) {
|
||||||
|
a := Facts{}.Block(ref)
|
||||||
|
c := Facts{}.Block(ref.Add(time.Hour))
|
||||||
|
if a == c {
|
||||||
|
t.Errorf("block did not change with the clock:\n%s", a)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// She may only offer what this deployment actually has.
|
||||||
|
func TestCapabilitiesAreConfigGated(t *testing.T) {
|
||||||
|
bare := Facts{}.Block(ref)
|
||||||
|
for _, want := range []string{"напоминания", "заметки", "календарь"} {
|
||||||
|
if !strings.Contains(bare, want) {
|
||||||
|
t.Errorf("block missing always-on capability %q:\n%s", want, bare)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, unwanted := range []string{"погоду", "телеграм", "команды"} {
|
||||||
|
if strings.Contains(bare, unwanted) {
|
||||||
|
t.Errorf("block offers unconfigured %q:\n%s", unwanted, bare)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
full := Facts{Weather: true, Telegram: true, Tools: true}.Block(ref)
|
||||||
|
for _, want := range []string{"погоду", "телеграм", "команды"} {
|
||||||
|
if !strings.Contains(full, want) {
|
||||||
|
t.Errorf("block missing configured capability %q:\n%s", want, full)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestPrependNilIsSafe(t *testing.T) {
|
||||||
|
if got := Prepend(nil, "PROMPT"); got != "PROMPT" {
|
||||||
|
t.Errorf("Prepend(nil) = %q", got)
|
||||||
|
}
|
||||||
|
if got := Prepend(func() string { return " " }, "PROMPT"); got != "PROMPT" {
|
||||||
|
t.Errorf("Prepend(blank) = %q", got)
|
||||||
|
}
|
||||||
|
if got := Prepend(func() string { return "CTX" }, "PROMPT"); got != "CTX\n\nPROMPT" {
|
||||||
|
t.Errorf("Prepend = %q", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
package phraser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"errors"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// A reply that starts a JSON object and never finishes it is a failed
|
||||||
|
// generation, not a reply. Before this, the parser returned ("", "") for these
|
||||||
|
// and every caller then shipped the raw fragment as the thing Maven said. A
|
||||||
|
// real run produced replies of literally "{" and "{\n \"".
|
||||||
|
func TestParseResponseMoodRejectsUnfinishedJSON(t *testing.T) {
|
||||||
|
for _, raw := range []string{
|
||||||
|
`{`,
|
||||||
|
"{\n \"",
|
||||||
|
`{"response": "неполн`,
|
||||||
|
`{"response": "текст", "mood":`,
|
||||||
|
} {
|
||||||
|
text, mood, err := parseResponseMood(raw)
|
||||||
|
if !errors.Is(err, errBrokenJSON) {
|
||||||
|
t.Errorf("parseResponseMood(%q) err = %v, want errBrokenJSON", raw, err)
|
||||||
|
}
|
||||||
|
if text != "" || mood != "" {
|
||||||
|
t.Errorf("parseResponseMood(%q) leaked %q/%q — a fragment must never come back as a reply", raw, text, mood)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bare prose is still fine. Small models sometimes answer without any JSON at
|
||||||
|
// all, and that reply is usable — so the new error must not swallow it.
|
||||||
|
func TestParseResponseMoodAllowsBareProse(t *testing.T) {
|
||||||
|
for _, raw := range []string{
|
||||||
|
"норм, а ты как?",
|
||||||
|
"вот что я нашла: ключ у соседа",
|
||||||
|
} {
|
||||||
|
text, mood, err := parseResponseMood(raw)
|
||||||
|
if err != nil {
|
||||||
|
t.Errorf("parseResponseMood(%q) err = %v, want nil", raw, err)
|
||||||
|
}
|
||||||
|
// No JSON means no fields; the caller ships raw as-is.
|
||||||
|
if text != "" || mood != "" {
|
||||||
|
t.Errorf("parseResponseMood(%q) = %q/%q, want empty", raw, text, mood)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The measured failure: the model wants more than 400 characters and the old
|
||||||
|
// grammar cut it off mid-word. Guards the bound against being tightened back.
|
||||||
|
func TestGrammarStringBoundHasRoomForARealAnswer(t *testing.T) {
|
||||||
|
if !strings.Contains(responseGrammar, "{0,1000}") {
|
||||||
|
t.Error("grammar string bound is not 1000; 400 truncated real replies mid-word (see the comment on responseGrammar)")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
package phraser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// Every phrasing prompt must carry the shared context block. This is the
|
||||||
|
// regression guard for the bug that started this: the "ты" rule reached only
|
||||||
|
// two of the five prompts because each prompt had its own copy of the rules.
|
||||||
|
func TestEveryPromptCarriesTheContextBlock(t *testing.T) {
|
||||||
|
block := func() string { return "CTXBLOCK" }
|
||||||
|
p := &LLMPhraser{cfg: Config{ContextBlock: block}}
|
||||||
|
|
||||||
|
prompts := map[string]string{
|
||||||
|
"nudge": p.systemPrompt(),
|
||||||
|
"query": p.querySystemPrompt(),
|
||||||
|
"chat": chatSystemPrompt(block),
|
||||||
|
}
|
||||||
|
for name, got := range prompts {
|
||||||
|
if !strings.HasPrefix(got, "CTXBLOCK\n\n") {
|
||||||
|
t.Errorf("%s prompt does not start with the context block:\n%s", name, got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Without a block the prompts are unchanged — the stub and test paths pass nil.
|
||||||
|
func TestPromptsWithoutBlockAreUnchanged(t *testing.T) {
|
||||||
|
p := &LLMPhraser{}
|
||||||
|
if p.systemPrompt() != nudgeSystem {
|
||||||
|
t.Errorf("nudge prompt changed with no block set")
|
||||||
|
}
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user