Add golden-audio STT tests against real whisper.cpp (#288)
Four committed WAV fixtures go through the real whisper.cpp binding in cmd/mavsttd, so a wrong model, a wrong language hint, a broken resample or a regressed silence gate fails `make test` instead of surfacing as Maven mishearing him. The fixtures are piper-synthesised, not recorded: scripts/gen-stt-fixtures.sh drives the vendored piper with the ru_RU-irina voice Maven already speaks with, so nothing of the owner's voice is committed and every fixture is reproducible. 360K total for three Russian clips and one English. Matching is tolerant on purpose. Golden transcripts move with the model, so each case asserts intent-carrying keywords (prefix match, so Russian inflection does not fail it) plus a word error rate ceiling, not an exact string. The matcher is unit-tested on its own and needs no model. TestGoldenAudioTranscription skips when models/stt/ggml-small.bin is absent, so `make test` still passes on a box without models. TestGoldenFixturesAreCanonical runs everywhere and checks the WAVs are 16k mono s16le and would clear mavsttd's own silence gate.
This commit is contained in:
@@ -16,7 +16,7 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
|
||||
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
||||
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
||||
|
||||
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||
.PHONY: stt-fixtures test-stt-golden all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||
|
||||
all: build
|
||||
|
||||
@@ -139,6 +139,17 @@ eval-models:
|
||||
MAVEN_LLM_URL="$(MAVEN_LLM_URL)" $(GO) test -v -count=1 -timeout 60m \
|
||||
-run TestLLMRouterBaseline ./internal/router/eval/
|
||||
|
||||
# stt-fixtures — regenerate the golden STT audio in cmd/mavsttd/testdata from
|
||||
# the piper voices (#288). The committed WAVs are synthesised, never recorded,
|
||||
# so this is the only way they should ever change. TestGoldenAudioTranscription
|
||||
# then scores them against ggml-small; it self-skips when the model is absent.
|
||||
stt-fixtures:
|
||||
./scripts/gen-stt-fixtures.sh
|
||||
|
||||
test-stt-golden:
|
||||
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||
$(GO) test -v -count=1 -run TestGolden ./cmd/mavsttd/
|
||||
|
||||
run-stt: build-stt
|
||||
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
||||
|
||||
@@ -0,0 +1,323 @@
|
||||
package main
|
||||
|
||||
// Golden-audio STT tests (Vikunja #288).
|
||||
//
|
||||
// These push real audio through the real whisper.cpp binding, so a bad model
|
||||
// path, a wrong language hint, a broken resample or a regressed silence gate
|
||||
// is caught by `make test` rather than by the owner talking to a daemon that
|
||||
// mishears him.
|
||||
//
|
||||
// The fixtures are piper-synthesised, not recorded — see
|
||||
// scripts/gen-stt-fixtures.sh. Nothing of the owner's voice is committed, and
|
||||
// any fixture can be rebuilt from the script plus a voice model.
|
||||
//
|
||||
// Matching is deliberately tolerant. Golden transcripts are model-dependent:
|
||||
// swapping ggml-small for a different whisper build moves punctuation, casing
|
||||
// and the odd word ending, and an exact-string assertion would turn every
|
||||
// model swap into a fixture rewrite. Each case therefore asserts two things —
|
||||
// the words that carry the intent are present, and the word error rate
|
||||
// against the reference stays under a per-case ceiling.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
"github.com/kami/maven/internal/worker"
|
||||
)
|
||||
|
||||
// goldenModelPath — the whisper model the golden tests run against. Same file
|
||||
// the Makefile's run-stt target uses. Overridable so a box that keeps its
|
||||
// models elsewhere can still run these.
|
||||
func goldenModelPath() string {
|
||||
if p := os.Getenv("MAVEN_WHISPER_MODEL"); p != "" {
|
||||
return p
|
||||
}
|
||||
return filepath.Join("..", "..", "models", "stt", "ggml-small.bin")
|
||||
}
|
||||
|
||||
type goldenCase struct {
|
||||
Name string `json:"name"`
|
||||
WAV string `json:"wav"`
|
||||
Lang string `json:"lang"`
|
||||
Text string `json:"text"`
|
||||
Keywords []string `json:"keywords"`
|
||||
MaxWER float64 `json:"max_wer"`
|
||||
}
|
||||
|
||||
type goldenManifest struct {
|
||||
Cases []goldenCase `json:"cases"`
|
||||
}
|
||||
|
||||
func loadGoldenManifest(t *testing.T) goldenManifest {
|
||||
t.Helper()
|
||||
raw, err := os.ReadFile(filepath.Join("testdata", "golden_v1.json"))
|
||||
if err != nil {
|
||||
t.Fatalf("read golden manifest: %v", err)
|
||||
}
|
||||
var m goldenManifest
|
||||
if err := json.Unmarshal(raw, &m); err != nil {
|
||||
t.Fatalf("parse golden manifest: %v", err)
|
||||
}
|
||||
if len(m.Cases) == 0 {
|
||||
t.Fatal("golden manifest has no cases")
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// normalizeTranscript lowercases, drops punctuation, folds the Russian ё onto
|
||||
// е (whisper is inconsistent about it and the router does not care), and
|
||||
// collapses whitespace. Everything the comparison does happens on this form.
|
||||
func normalizeTranscript(s string) []string {
|
||||
var b strings.Builder
|
||||
for _, r := range strings.ToLower(s) {
|
||||
switch {
|
||||
case r == 'ё':
|
||||
b.WriteRune('е')
|
||||
case unicode.IsLetter(r) || unicode.IsDigit(r):
|
||||
b.WriteRune(r)
|
||||
default:
|
||||
b.WriteRune(' ')
|
||||
}
|
||||
}
|
||||
return strings.Fields(b.String())
|
||||
}
|
||||
|
||||
// wordErrorRate is the Levenshtein distance between two word sequences,
|
||||
// divided by the length of the reference. 0 means identical; it can exceed 1
|
||||
// when the hypothesis is much longer than the reference.
|
||||
func wordErrorRate(ref, hyp []string) float64 {
|
||||
if len(ref) == 0 {
|
||||
if len(hyp) == 0 {
|
||||
return 0
|
||||
}
|
||||
return 1
|
||||
}
|
||||
prev := make([]int, len(hyp)+1)
|
||||
cur := make([]int, len(hyp)+1)
|
||||
for j := range prev {
|
||||
prev[j] = j
|
||||
}
|
||||
for i := 1; i <= len(ref); i++ {
|
||||
cur[0] = i
|
||||
for j := 1; j <= len(hyp); j++ {
|
||||
cost := 1
|
||||
if ref[i-1] == hyp[j-1] {
|
||||
cost = 0
|
||||
}
|
||||
cur[j] = min(prev[j]+1, min(cur[j-1]+1, prev[j-1]+cost))
|
||||
}
|
||||
prev, cur = cur, prev
|
||||
}
|
||||
return float64(prev[len(hyp)]) / float64(len(ref))
|
||||
}
|
||||
|
||||
// missingKeywords returns the keywords absent from the hypothesis. A keyword
|
||||
// matches on prefix, so a different case ending ("воды" vs "воду") does not
|
||||
// fail the assertion — the router's stage-0 grammar is stem-shaped too.
|
||||
func missingKeywords(keywords []string, hyp []string) []string {
|
||||
var missing []string
|
||||
for _, kw := range keywords {
|
||||
want := normalizeTranscript(kw)
|
||||
if len(want) == 0 {
|
||||
continue
|
||||
}
|
||||
if !containsSeq(hyp, want) {
|
||||
missing = append(missing, kw)
|
||||
}
|
||||
}
|
||||
return missing
|
||||
}
|
||||
|
||||
func containsSeq(hyp, want []string) bool {
|
||||
for i := 0; i+len(want) <= len(hyp); i++ {
|
||||
ok := true
|
||||
for j, w := range want {
|
||||
// Prefix match, so inflection differences pass but
|
||||
// distinct words do not.
|
||||
if !looseWordMatch(hyp[i+j], w) {
|
||||
ok = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func looseWordMatch(got, want string) bool {
|
||||
if got == want {
|
||||
return true
|
||||
}
|
||||
g, w := []rune(got), []rune(want)
|
||||
n := len(w) - 1
|
||||
if len(w) > 6 {
|
||||
n = len(w) - 2
|
||||
}
|
||||
// Words of three runes or fewer have no room for a safe prefix: require
|
||||
// an exact match rather than letting "час" pass for "часть".
|
||||
if n < 3 || len(g) < n {
|
||||
return false
|
||||
}
|
||||
return string(g[:n]) == string(w[:n])
|
||||
}
|
||||
|
||||
// --- the model-backed test -------------------------------------------------
|
||||
|
||||
func TestGoldenAudioTranscription(t *testing.T) {
|
||||
m := loadGoldenManifest(t)
|
||||
|
||||
model := goldenModelPath()
|
||||
if _, err := os.Stat(model); err != nil {
|
||||
t.Skipf("whisper model %s absent (%v) — set MAVEN_WHISPER_MODEL or see AGENTS.md", model, err)
|
||||
}
|
||||
|
||||
// Same gate thresholds as mavsttd's defaults, so a regression in the
|
||||
// silence gate shows up here as an empty transcript.
|
||||
h, err := newWhisperHandler(model, 300, 0.01)
|
||||
if err != nil {
|
||||
t.Fatalf("load whisper model %s: %v", model, err)
|
||||
}
|
||||
defer h.Close()
|
||||
|
||||
for _, c := range m.Cases {
|
||||
t.Run(c.Name, func(t *testing.T) {
|
||||
path := filepath.Join("testdata", c.WAV)
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Skipf("fixture %s absent (%v) — run scripts/gen-stt-fixtures.sh", path, err)
|
||||
}
|
||||
format, pcm, err := audio.PCMFromWAV(raw)
|
||||
if err != nil {
|
||||
t.Fatalf("%s is not canonical 16k mono PCM: %v", path, err)
|
||||
}
|
||||
|
||||
resp, err := h.Transcribe(context.Background(), worker.TranscribeReq{
|
||||
Audio: audio.Audio{Format: format, Bytes: pcm},
|
||||
Lang: c.Lang,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("transcribe %s: %v", c.WAV, err)
|
||||
}
|
||||
t.Logf("%s → %q (confidence %.3f)", c.WAV, resp.Text, resp.Confidence)
|
||||
|
||||
if strings.TrimSpace(resp.Text) == "" {
|
||||
t.Fatalf("%s transcribed to empty text — the silence gate ate real speech", c.WAV)
|
||||
}
|
||||
if resp.Confidence <= 0 {
|
||||
t.Errorf("%s: confidence %v, want > 0", c.WAV, resp.Confidence)
|
||||
}
|
||||
|
||||
hyp := normalizeTranscript(resp.Text)
|
||||
ref := normalizeTranscript(c.Text)
|
||||
|
||||
if missing := missingKeywords(c.Keywords, hyp); len(missing) > 0 {
|
||||
t.Errorf("%s: missing keywords %v in %q", c.WAV, missing, resp.Text)
|
||||
}
|
||||
if wer := wordErrorRate(ref, hyp); wer > c.MaxWER {
|
||||
t.Errorf("%s: WER %.2f > %.2f\n want: %q\n got: %q", c.WAV, wer, c.MaxWER, c.Text, resp.Text)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoldenFixturesAreCanonical checks the committed audio without needing a
|
||||
// model, so a fixture regenerated at the wrong sample rate fails on every box.
|
||||
func TestGoldenFixturesAreCanonical(t *testing.T) {
|
||||
m := loadGoldenManifest(t)
|
||||
for _, c := range m.Cases {
|
||||
path := filepath.Join("testdata", c.WAV)
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Errorf("fixture %s missing: %v", path, err)
|
||||
continue
|
||||
}
|
||||
format, pcm, err := audio.PCMFromWAV(raw)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", path, err)
|
||||
continue
|
||||
}
|
||||
if !format.IsValid() {
|
||||
t.Errorf("%s: format %+v is not canonical", path, format)
|
||||
}
|
||||
a := audio.Audio{Format: format, Bytes: pcm}
|
||||
if d := a.Duration(); d < 0.5 || d > 10 {
|
||||
t.Errorf("%s: duration %.2fs outside the sane 0.5–10s fixture range", path, d)
|
||||
}
|
||||
// The fixture must clear mavsttd's own silence gate, otherwise the
|
||||
// model test below would be asserting on a gated empty string.
|
||||
if reason := gateReason(pcmToF32(pcm), whisperSampleRate, 300, 0.01); reason != "" {
|
||||
t.Errorf("%s: would be gated as %s", path, reason)
|
||||
}
|
||||
if len(c.Keywords) == 0 {
|
||||
t.Errorf("%s: manifest case has no keywords", c.Name)
|
||||
}
|
||||
if c.MaxWER <= 0 || c.MaxWER > 1 {
|
||||
t.Errorf("%s: max_wer %v outside (0,1]", c.Name, c.MaxWER)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func pcmToF32(b []byte) []float32 {
|
||||
out := make([]float32, len(b)/2)
|
||||
for i := range out {
|
||||
s := int16(b[i*2]) | int16(b[i*2+1])<<8
|
||||
out[i] = float32(s) / 32768.0
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// --- matcher unit tests (no model, no fixtures) ----------------------------
|
||||
|
||||
func TestNormalizeTranscript(t *testing.T) {
|
||||
got := normalizeTranscript(" Ещё, Раз... ")
|
||||
want := []string{"еще", "раз"}
|
||||
if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] {
|
||||
t.Fatalf("normalizeTranscript = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWordErrorRate(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
ref, hyp string
|
||||
want float64
|
||||
}{
|
||||
{"identical", "напомни мне через час", "Напомни мне через час.", 0},
|
||||
{"one substitution", "напомни мне через час", "напомни мне через день", 0.25},
|
||||
{"one deletion", "напомни мне через час", "напомни мне час", 0.25},
|
||||
{"empty hypothesis", "напомни мне", "", 1},
|
||||
{"both empty", "", "", 0},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := wordErrorRate(normalizeTranscript(c.ref), normalizeTranscript(c.hyp))
|
||||
if got != c.want {
|
||||
t.Fatalf("WER = %v, want %v", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingKeywords(t *testing.T) {
|
||||
hyp := normalizeTranscript("Отметь, что я выпил воду.")
|
||||
if got := missingKeywords([]string{"воды", "отметь"}, hyp); len(got) != 0 {
|
||||
t.Fatalf("missingKeywords = %v, want none (inflection must not fail the match)", got)
|
||||
}
|
||||
if got := missingKeywords([]string{"календарю"}, hyp); len(got) != 1 {
|
||||
t.Fatalf("missingKeywords = %v, want the absent keyword reported", got)
|
||||
}
|
||||
// A short word must match exactly — no 4-rune prefix shortcut that would
|
||||
// let "час" pass for "часть".
|
||||
hyp2 := normalizeTranscript("через час")
|
||||
if got := missingKeywords([]string{"часть"}, hyp2); len(got) != 1 {
|
||||
t.Fatalf("missingKeywords = %v, want %q reported missing", got, "часть")
|
||||
}
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
Vendored
+37
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"note": "Golden STT fixtures. Audio is piper-synthesised, not recorded — see scripts/gen-stt-fixtures.sh. Regenerate with that script; do not hand-edit `wav`.",
|
||||
"cases": [
|
||||
{
|
||||
"name": "ru_reminder",
|
||||
"wav": "ru_reminder.wav",
|
||||
"lang": "ru",
|
||||
"text": "напомни мне через час позвонить маме",
|
||||
"keywords": ["напомни", "час", "позвонить"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "ru_fact",
|
||||
"wav": "ru_fact.wav",
|
||||
"lang": "ru",
|
||||
"text": "отметь что я выпил воды",
|
||||
"keywords": ["отметь", "воды"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "ru_query",
|
||||
"wav": "ru_query.wav",
|
||||
"lang": "ru",
|
||||
"text": "что у меня сегодня по календарю",
|
||||
"keywords": ["сегодня", "календарю"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "en_act",
|
||||
"wav": "en_act.wav",
|
||||
"lang": "en",
|
||||
"text": "restart the web server and check the disk space",
|
||||
"keywords": ["restart", "server", "disk"],
|
||||
"max_wer": 0.34
|
||||
}
|
||||
]
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
Executable
+67
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env bash
|
||||
# gen-stt-fixtures.sh — regenerate the golden STT audio fixtures.
|
||||
#
|
||||
# The fixtures in cmd/mavsttd/testdata/*.wav are SYNTHESISED, not recorded.
|
||||
# They come out of the same piper voices maven speaks with, so nothing of the
|
||||
# owner's voice is committed and every fixture is reproducible from this
|
||||
# script plus the voice model. They are also small: 16 kHz mono s16le, a
|
||||
# couple of seconds each.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/gen-stt-fixtures.sh
|
||||
#
|
||||
# Voices are picked up from, in order, $PIPER_VOICE_RU / $PIPER_VOICE_EN, then
|
||||
# the repo's models/tts, then ~/esp-server/voices. The English voice is not
|
||||
# vendored; if it is missing the English fixture is skipped and the existing
|
||||
# one is left alone.
|
||||
set -euo pipefail
|
||||
|
||||
root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
out="$root/cmd/mavsttd/testdata"
|
||||
piper="${PIPER_BIN:-$root/deps/piper/piper}"
|
||||
espeak="${PIPER_ESPEAK:-$root/deps/piper/espeak-ng-data}"
|
||||
|
||||
pick_voice() {
|
||||
for c in "$@"; do
|
||||
[ -f "$c" ] && { echo "$c"; return 0; }
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
ru="$(pick_voice "${PIPER_VOICE_RU:-}" "$root/models/tts/ru_RU-irina-medium.onnx" "$HOME/esp-server/voices/ru_RU-irina-medium.onnx")" || {
|
||||
echo "no russian piper voice found" >&2
|
||||
exit 1
|
||||
}
|
||||
en="$(pick_voice "${PIPER_VOICE_EN:-}" "$root/models/tts/en_US-lessac-medium.onnx" "$HOME/esp-server/voices/en_US-lessac-medium.onnx")" || en=""
|
||||
|
||||
# synth <voice> <out.wav> <text>
|
||||
# piper emits raw 22050 Hz s16le on stdout; ffmpeg resamples to the canonical
|
||||
# 16 kHz mono and writes a plain 44-byte-header WAV (-fflags bitexact keeps
|
||||
# ffmpeg's encoder LIST chunk out, so the bytes are stable across ffmpeg
|
||||
# builds and internal/audio.PCMFromWAV reads them without scanning).
|
||||
synth() {
|
||||
local voice="$1" dest="$2" text="$3"
|
||||
printf '%s' "$text" | LD_LIBRARY_PATH="$(dirname "$piper")" "$piper" \
|
||||
--model "$voice" --config "$voice.json" \
|
||||
--espeak_data "$espeak" --output_raw --quiet |
|
||||
ffmpeg -hide_banner -loglevel error -y \
|
||||
-f s16le -ar 22050 -ac 1 -i - \
|
||||
-af "adelay=200,apad=pad_dur=0.2" \
|
||||
-ar 16000 -ac 1 -c:a pcm_s16le -fflags bitexact "$dest"
|
||||
echo "wrote $dest ($(stat -c%s "$dest") bytes)"
|
||||
}
|
||||
|
||||
synth "$ru" "$out/ru_reminder.wav" "Напомни мне через час позвонить маме."
|
||||
synth "$ru" "$out/ru_fact.wav" "Отметь, что я выпил воды."
|
||||
synth "$ru" "$out/ru_query.wav" "Что у меня сегодня по календарю?"
|
||||
|
||||
if [ -n "$en" ]; then
|
||||
# Keep the English line free of words piper spells out letter by letter —
|
||||
# "nginx" comes out of lessac as "engine X", which is a TTS artefact and
|
||||
# would make the fixture assert on the wrong thing.
|
||||
synth "$en" "$out/en_act.wav" "Restart the web server and check the disk space."
|
||||
else
|
||||
echo "no english piper voice found — skipping en_act.wav" >&2
|
||||
fi
|
||||
|
||||
echo "fixtures regenerated; expected transcripts live in $out/golden_v1.json"
|
||||
Reference in New Issue
Block a user