Golden-audio STT tests against real whisper.cpp (#288) #75

Closed
claude wants to merge 1 commits from overnight/stt-golden-audio into overnight/senses-speaker
8 changed files with 439 additions and 1 deletions
+12 -1
View File
@@ -16,7 +16,7 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
.PHONY: stt-fixtures test-stt-golden all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
all: build
@@ -139,6 +139,17 @@ eval-models:
MAVEN_LLM_URL="$(MAVEN_LLM_URL)" $(GO) test -v -count=1 -timeout 60m \
-run TestLLMRouterBaseline ./internal/router/eval/
# stt-fixtures — regenerate the golden STT audio in cmd/mavsttd/testdata from
# the piper voices (#288). The committed WAVs are synthesised, never recorded,
# so this is the only way they should ever change. TestGoldenAudioTranscription
# then scores them against ggml-small; it self-skips when the model is absent.
stt-fixtures:
./scripts/gen-stt-fixtures.sh
test-stt-golden:
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
$(GO) test -v -count=1 -run TestGolden ./cmd/mavsttd/
run-stt: build-stt
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
+323
View File
@@ -0,0 +1,323 @@
package main
// Golden-audio STT tests (Vikunja #288).
//
// These push real audio through the real whisper.cpp binding, so a bad model
// path, a wrong language hint, a broken resample or a regressed silence gate
// is caught by `make test` rather than by the owner talking to a daemon that
// mishears him.
//
// The fixtures are piper-synthesised, not recorded — see
// scripts/gen-stt-fixtures.sh. Nothing of the owner's voice is committed, and
// any fixture can be rebuilt from the script plus a voice model.
//
// Matching is deliberately tolerant. Golden transcripts are model-dependent:
// swapping ggml-small for a different whisper build moves punctuation, casing
// and the odd word ending, and an exact-string assertion would turn every
// model swap into a fixture rewrite. Each case therefore asserts two things —
// the words that carry the intent are present, and the word error rate
// against the reference stays under a per-case ceiling.
import (
"context"
"encoding/json"
"os"
"path/filepath"
"strings"
"testing"
"unicode"
"github.com/kami/maven/internal/audio"
"github.com/kami/maven/internal/worker"
)
// goldenModelPath — the whisper model the golden tests run against. Same file
// the Makefile's run-stt target uses. Overridable so a box that keeps its
// models elsewhere can still run these.
func goldenModelPath() string {
if p := os.Getenv("MAVEN_WHISPER_MODEL"); p != "" {
return p
}
return filepath.Join("..", "..", "models", "stt", "ggml-small.bin")
}
type goldenCase struct {
Name string `json:"name"`
WAV string `json:"wav"`
Lang string `json:"lang"`
Text string `json:"text"`
Keywords []string `json:"keywords"`
MaxWER float64 `json:"max_wer"`
}
type goldenManifest struct {
Cases []goldenCase `json:"cases"`
}
func loadGoldenManifest(t *testing.T) goldenManifest {
t.Helper()
raw, err := os.ReadFile(filepath.Join("testdata", "golden_v1.json"))
if err != nil {
t.Fatalf("read golden manifest: %v", err)
}
var m goldenManifest
if err := json.Unmarshal(raw, &m); err != nil {
t.Fatalf("parse golden manifest: %v", err)
}
if len(m.Cases) == 0 {
t.Fatal("golden manifest has no cases")
}
return m
}
// normalizeTranscript lowercases, drops punctuation, folds the Russian ё onto
// е (whisper is inconsistent about it and the router does not care), and
// collapses whitespace. Everything the comparison does happens on this form.
func normalizeTranscript(s string) []string {
var b strings.Builder
for _, r := range strings.ToLower(s) {
switch {
case r == 'ё':
b.WriteRune('е')
case unicode.IsLetter(r) || unicode.IsDigit(r):
b.WriteRune(r)
default:
b.WriteRune(' ')
}
}
return strings.Fields(b.String())
}
// wordErrorRate is the Levenshtein distance between two word sequences,
// divided by the length of the reference. 0 means identical; it can exceed 1
// when the hypothesis is much longer than the reference.
func wordErrorRate(ref, hyp []string) float64 {
if len(ref) == 0 {
if len(hyp) == 0 {
return 0
}
return 1
}
prev := make([]int, len(hyp)+1)
cur := make([]int, len(hyp)+1)
for j := range prev {
prev[j] = j
}
for i := 1; i <= len(ref); i++ {
cur[0] = i
for j := 1; j <= len(hyp); j++ {
cost := 1
if ref[i-1] == hyp[j-1] {
cost = 0
}
cur[j] = min(prev[j]+1, min(cur[j-1]+1, prev[j-1]+cost))
}
prev, cur = cur, prev
}
return float64(prev[len(hyp)]) / float64(len(ref))
}
// missingKeywords returns the keywords absent from the hypothesis. A keyword
// matches on prefix, so a different case ending ("воды" vs "воду") does not
// fail the assertion — the router's stage-0 grammar is stem-shaped too.
func missingKeywords(keywords []string, hyp []string) []string {
var missing []string
for _, kw := range keywords {
want := normalizeTranscript(kw)
if len(want) == 0 {
continue
}
if !containsSeq(hyp, want) {
missing = append(missing, kw)
}
}
return missing
}
func containsSeq(hyp, want []string) bool {
for i := 0; i+len(want) <= len(hyp); i++ {
ok := true
for j, w := range want {
// Prefix match, so inflection differences pass but
// distinct words do not.
if !looseWordMatch(hyp[i+j], w) {
ok = false
break
}
}
if ok {
return true
}
}
return false
}
func looseWordMatch(got, want string) bool {
if got == want {
return true
}
g, w := []rune(got), []rune(want)
n := len(w) - 1
if len(w) > 6 {
n = len(w) - 2
}
// Words of three runes or fewer have no room for a safe prefix: require
// an exact match rather than letting "час" pass for "часть".
if n < 3 || len(g) < n {
return false
}
return string(g[:n]) == string(w[:n])
}
// --- the model-backed test -------------------------------------------------
func TestGoldenAudioTranscription(t *testing.T) {
m := loadGoldenManifest(t)
model := goldenModelPath()
if _, err := os.Stat(model); err != nil {
t.Skipf("whisper model %s absent (%v) — set MAVEN_WHISPER_MODEL or see AGENTS.md", model, err)
}
// Same gate thresholds as mavsttd's defaults, so a regression in the
// silence gate shows up here as an empty transcript.
h, err := newWhisperHandler(model, 300, 0.01)
if err != nil {
t.Fatalf("load whisper model %s: %v", model, err)
}
defer h.Close()
for _, c := range m.Cases {
t.Run(c.Name, func(t *testing.T) {
path := filepath.Join("testdata", c.WAV)
raw, err := os.ReadFile(path)
if err != nil {
t.Skipf("fixture %s absent (%v) — run scripts/gen-stt-fixtures.sh", path, err)
}
format, pcm, err := audio.PCMFromWAV(raw)
if err != nil {
t.Fatalf("%s is not canonical 16k mono PCM: %v", path, err)
}
resp, err := h.Transcribe(context.Background(), worker.TranscribeReq{
Audio: audio.Audio{Format: format, Bytes: pcm},
Lang: c.Lang,
})
if err != nil {
t.Fatalf("transcribe %s: %v", c.WAV, err)
}
t.Logf("%s → %q (confidence %.3f)", c.WAV, resp.Text, resp.Confidence)
if strings.TrimSpace(resp.Text) == "" {
t.Fatalf("%s transcribed to empty text — the silence gate ate real speech", c.WAV)
}
if resp.Confidence <= 0 {
t.Errorf("%s: confidence %v, want > 0", c.WAV, resp.Confidence)
}
hyp := normalizeTranscript(resp.Text)
ref := normalizeTranscript(c.Text)
if missing := missingKeywords(c.Keywords, hyp); len(missing) > 0 {
t.Errorf("%s: missing keywords %v in %q", c.WAV, missing, resp.Text)
}
if wer := wordErrorRate(ref, hyp); wer > c.MaxWER {
t.Errorf("%s: WER %.2f > %.2f\n want: %q\n got: %q", c.WAV, wer, c.MaxWER, c.Text, resp.Text)
}
})
}
}
// TestGoldenFixturesAreCanonical checks the committed audio without needing a
// model, so a fixture regenerated at the wrong sample rate fails on every box.
func TestGoldenFixturesAreCanonical(t *testing.T) {
m := loadGoldenManifest(t)
for _, c := range m.Cases {
path := filepath.Join("testdata", c.WAV)
raw, err := os.ReadFile(path)
if err != nil {
t.Errorf("fixture %s missing: %v", path, err)
continue
}
format, pcm, err := audio.PCMFromWAV(raw)
if err != nil {
t.Errorf("%s: %v", path, err)
continue
}
if !format.IsValid() {
t.Errorf("%s: format %+v is not canonical", path, format)
}
a := audio.Audio{Format: format, Bytes: pcm}
if d := a.Duration(); d < 0.5 || d > 10 {
t.Errorf("%s: duration %.2fs outside the sane 0.510s fixture range", path, d)
}
// The fixture must clear mavsttd's own silence gate, otherwise the
// model test below would be asserting on a gated empty string.
if reason := gateReason(pcmToF32(pcm), whisperSampleRate, 300, 0.01); reason != "" {
t.Errorf("%s: would be gated as %s", path, reason)
}
if len(c.Keywords) == 0 {
t.Errorf("%s: manifest case has no keywords", c.Name)
}
if c.MaxWER <= 0 || c.MaxWER > 1 {
t.Errorf("%s: max_wer %v outside (0,1]", c.Name, c.MaxWER)
}
}
}
func pcmToF32(b []byte) []float32 {
out := make([]float32, len(b)/2)
for i := range out {
s := int16(b[i*2]) | int16(b[i*2+1])<<8
out[i] = float32(s) / 32768.0
}
return out
}
// --- matcher unit tests (no model, no fixtures) ----------------------------
func TestNormalizeTranscript(t *testing.T) {
got := normalizeTranscript(" Ещё, Раз... ")
want := []string{"еще", "раз"}
if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] {
t.Fatalf("normalizeTranscript = %v, want %v", got, want)
}
}
func TestWordErrorRate(t *testing.T) {
cases := []struct {
name string
ref, hyp string
want float64
}{
{"identical", "напомни мне через час", "Напомни мне через час.", 0},
{"one substitution", "напомни мне через час", "напомни мне через день", 0.25},
{"one deletion", "напомни мне через час", "напомни мне час", 0.25},
{"empty hypothesis", "напомни мне", "", 1},
{"both empty", "", "", 0},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
got := wordErrorRate(normalizeTranscript(c.ref), normalizeTranscript(c.hyp))
if got != c.want {
t.Fatalf("WER = %v, want %v", got, c.want)
}
})
}
}
func TestMissingKeywords(t *testing.T) {
hyp := normalizeTranscript("Отметь, что я выпил воду.")
if got := missingKeywords([]string{"воды", "отметь"}, hyp); len(got) != 0 {
t.Fatalf("missingKeywords = %v, want none (inflection must not fail the match)", got)
}
if got := missingKeywords([]string{"календарю"}, hyp); len(got) != 1 {
t.Fatalf("missingKeywords = %v, want the absent keyword reported", got)
}
// A short word must match exactly — no 4-rune prefix shortcut that would
// let "час" pass for "часть".
hyp2 := normalizeTranscript("через час")
if got := missingKeywords([]string{"часть"}, hyp2); len(got) != 1 {
t.Fatalf("missingKeywords = %v, want %q reported missing", got, "часть")
}
}
BIN
View File
Binary file not shown.
+37
View File
@@ -0,0 +1,37 @@
{
"note": "Golden STT fixtures. Audio is piper-synthesised, not recorded — see scripts/gen-stt-fixtures.sh. Regenerate with that script; do not hand-edit `wav`.",
"cases": [
{
"name": "ru_reminder",
"wav": "ru_reminder.wav",
"lang": "ru",
"text": "напомни мне через час позвонить маме",
"keywords": ["напомни", "час", "позвонить"],
"max_wer": 0.34
},
{
"name": "ru_fact",
"wav": "ru_fact.wav",
"lang": "ru",
"text": "отметь что я выпил воды",
"keywords": ["отметь", "воды"],
"max_wer": 0.34
},
{
"name": "ru_query",
"wav": "ru_query.wav",
"lang": "ru",
"text": "что у меня сегодня по календарю",
"keywords": ["сегодня", "календарю"],
"max_wer": 0.34
},
{
"name": "en_act",
"wav": "en_act.wav",
"lang": "en",
"text": "restart the web server and check the disk space",
"keywords": ["restart", "server", "disk"],
"max_wer": 0.34
}
]
}
Binary file not shown.
Binary file not shown.
Binary file not shown.
+67
View File
@@ -0,0 +1,67 @@
#!/usr/bin/env bash
# gen-stt-fixtures.sh — regenerate the golden STT audio fixtures.
#
# The fixtures in cmd/mavsttd/testdata/*.wav are SYNTHESISED, not recorded.
# They come out of the same piper voices maven speaks with, so nothing of the
# owner's voice is committed and every fixture is reproducible from this
# script plus the voice model. They are also small: 16 kHz mono s16le, a
# couple of seconds each.
#
# Usage:
# scripts/gen-stt-fixtures.sh
#
# Voices are picked up from, in order, $PIPER_VOICE_RU / $PIPER_VOICE_EN, then
# the repo's models/tts, then ~/esp-server/voices. The English voice is not
# vendored; if it is missing the English fixture is skipped and the existing
# one is left alone.
set -euo pipefail
root="$(cd "$(dirname "$0")/.." && pwd)"
out="$root/cmd/mavsttd/testdata"
piper="${PIPER_BIN:-$root/deps/piper/piper}"
espeak="${PIPER_ESPEAK:-$root/deps/piper/espeak-ng-data}"
pick_voice() {
for c in "$@"; do
[ -f "$c" ] && { echo "$c"; return 0; }
done
return 1
}
ru="$(pick_voice "${PIPER_VOICE_RU:-}" "$root/models/tts/ru_RU-irina-medium.onnx" "$HOME/esp-server/voices/ru_RU-irina-medium.onnx")" || {
echo "no russian piper voice found" >&2
exit 1
}
en="$(pick_voice "${PIPER_VOICE_EN:-}" "$root/models/tts/en_US-lessac-medium.onnx" "$HOME/esp-server/voices/en_US-lessac-medium.onnx")" || en=""
# synth <voice> <out.wav> <text>
# piper emits raw 22050 Hz s16le on stdout; ffmpeg resamples to the canonical
# 16 kHz mono and writes a plain 44-byte-header WAV (-fflags bitexact keeps
# ffmpeg's encoder LIST chunk out, so the bytes are stable across ffmpeg
# builds and internal/audio.PCMFromWAV reads them without scanning).
synth() {
local voice="$1" dest="$2" text="$3"
printf '%s' "$text" | LD_LIBRARY_PATH="$(dirname "$piper")" "$piper" \
--model "$voice" --config "$voice.json" \
--espeak_data "$espeak" --output_raw --quiet |
ffmpeg -hide_banner -loglevel error -y \
-f s16le -ar 22050 -ac 1 -i - \
-af "adelay=200,apad=pad_dur=0.2" \
-ar 16000 -ac 1 -c:a pcm_s16le -fflags bitexact "$dest"
echo "wrote $dest ($(stat -c%s "$dest") bytes)"
}
synth "$ru" "$out/ru_reminder.wav" "Напомни мне через час позвонить маме."
synth "$ru" "$out/ru_fact.wav" "Отметь, что я выпил воды."
synth "$ru" "$out/ru_query.wav" "Что у меня сегодня по календарю?"
if [ -n "$en" ]; then
# Keep the English line free of words piper spells out letter by letter —
# "nginx" comes out of lessac as "engine X", which is a TTS artefact and
# would make the fixture assert on the wrong thing.
synth "$en" "$out/en_act.wav" "Restart the web server and check the disk space."
else
echo "no english piper voice found — skipping en_act.wav" >&2
fi
echo "fixtures regenerated; expected transcripts live in $out/golden_v1.json"