Golden-audio STT tests against real whisper.cpp (#288) #75
@@ -16,7 +16,7 @@ PIPER_BIN := $(shell pwd)/deps/piper/piper
|
||||
PIPER_MODEL := $(shell pwd)/models/tts/ru_RU-irina-medium.onnx
|
||||
PIPER_ESPEAK := $(shell pwd)/deps/piper/espeak-ng-data
|
||||
|
||||
.PHONY: all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||
.PHONY: stt-fixtures test-stt-golden all build build-stt build-tts build-daemon build-client build-waked build-web build-poll build-caldav clean test fmt-check vet run-stt run-tts run-web download-embedder deps-go eval-router eval-recall eval-phrasing eval-models
|
||||
|
||||
all: build
|
||||
|
||||
@@ -139,6 +139,17 @@ eval-models:
|
||||
MAVEN_LLM_URL="$(MAVEN_LLM_URL)" $(GO) test -v -count=1 -timeout 60m \
|
||||
-run TestLLMRouterBaseline ./internal/router/eval/
|
||||
|
||||
# stt-fixtures — regenerate the golden STT audio in cmd/mavsttd/testdata from
|
||||
# the piper voices (#288). The committed WAVs are synthesised, never recorded,
|
||||
# so this is the only way they should ever change. TestGoldenAudioTranscription
|
||||
# then scores them against ggml-small; it self-skips when the model is absent.
|
||||
stt-fixtures:
|
||||
./scripts/gen-stt-fixtures.sh
|
||||
|
||||
test-stt-golden:
|
||||
CGO_CFLAGS="$(CGO_CFLAGS)" CGO_LDFLAGS="$(CGO_LDFLAGS)" LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||
$(GO) test -v -count=1 -run TestGolden ./cmd/mavsttd/
|
||||
|
||||
run-stt: build-stt
|
||||
LD_LIBRARY_PATH="$(shell pwd)/deps/lib" \
|
||||
./mavsttd -socket /tmp/maven/stt.sock -model $(WHISPER_MODEL)
|
||||
|
||||
@@ -0,0 +1,323 @@
|
||||
package main
|
||||
|
||||
// Golden-audio STT tests (Vikunja #288).
|
||||
//
|
||||
// These push real audio through the real whisper.cpp binding, so a bad model
|
||||
// path, a wrong language hint, a broken resample or a regressed silence gate
|
||||
// is caught by `make test` rather than by the owner talking to a daemon that
|
||||
// mishears him.
|
||||
//
|
||||
// The fixtures are piper-synthesised, not recorded — see
|
||||
// scripts/gen-stt-fixtures.sh. Nothing of the owner's voice is committed, and
|
||||
// any fixture can be rebuilt from the script plus a voice model.
|
||||
//
|
||||
// Matching is deliberately tolerant. Golden transcripts are model-dependent:
|
||||
// swapping ggml-small for a different whisper build moves punctuation, casing
|
||||
// and the odd word ending, and an exact-string assertion would turn every
|
||||
// model swap into a fixture rewrite. Each case therefore asserts two things —
|
||||
// the words that carry the intent are present, and the word error rate
|
||||
// against the reference stays under a per-case ceiling.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode"
|
||||
|
||||
"github.com/kami/maven/internal/audio"
|
||||
"github.com/kami/maven/internal/worker"
|
||||
)
|
||||
|
||||
// goldenModelPath — the whisper model the golden tests run against. Same file
|
||||
// the Makefile's run-stt target uses. Overridable so a box that keeps its
|
||||
// models elsewhere can still run these.
|
||||
func goldenModelPath() string {
|
||||
if p := os.Getenv("MAVEN_WHISPER_MODEL"); p != "" {
|
||||
return p
|
||||
}
|
||||
return filepath.Join("..", "..", "models", "stt", "ggml-small.bin")
|
||||
}
|
||||
|
||||
type goldenCase struct {
|
||||
Name string `json:"name"`
|
||||
WAV string `json:"wav"`
|
||||
Lang string `json:"lang"`
|
||||
Text string `json:"text"`
|
||||
Keywords []string `json:"keywords"`
|
||||
MaxWER float64 `json:"max_wer"`
|
||||
}
|
||||
|
||||
type goldenManifest struct {
|
||||
Cases []goldenCase `json:"cases"`
|
||||
}
|
||||
|
||||
func loadGoldenManifest(t *testing.T) goldenManifest {
|
||||
t.Helper()
|
||||
raw, err := os.ReadFile(filepath.Join("testdata", "golden_v1.json"))
|
||||
if err != nil {
|
||||
t.Fatalf("read golden manifest: %v", err)
|
||||
}
|
||||
var m goldenManifest
|
||||
if err := json.Unmarshal(raw, &m); err != nil {
|
||||
t.Fatalf("parse golden manifest: %v", err)
|
||||
}
|
||||
if len(m.Cases) == 0 {
|
||||
t.Fatal("golden manifest has no cases")
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// normalizeTranscript lowercases, drops punctuation, folds the Russian ё onto
|
||||
// е (whisper is inconsistent about it and the router does not care), and
|
||||
// collapses whitespace. Everything the comparison does happens on this form.
|
||||
func normalizeTranscript(s string) []string {
|
||||
var b strings.Builder
|
||||
for _, r := range strings.ToLower(s) {
|
||||
switch {
|
||||
case r == 'ё':
|
||||
b.WriteRune('е')
|
||||
case unicode.IsLetter(r) || unicode.IsDigit(r):
|
||||
b.WriteRune(r)
|
||||
default:
|
||||
b.WriteRune(' ')
|
||||
}
|
||||
}
|
||||
return strings.Fields(b.String())
|
||||
}
|
||||
|
||||
// wordErrorRate is the Levenshtein distance between two word sequences,
|
||||
// divided by the length of the reference. 0 means identical; it can exceed 1
|
||||
// when the hypothesis is much longer than the reference.
|
||||
func wordErrorRate(ref, hyp []string) float64 {
|
||||
if len(ref) == 0 {
|
||||
if len(hyp) == 0 {
|
||||
return 0
|
||||
}
|
||||
return 1
|
||||
}
|
||||
prev := make([]int, len(hyp)+1)
|
||||
cur := make([]int, len(hyp)+1)
|
||||
for j := range prev {
|
||||
prev[j] = j
|
||||
}
|
||||
for i := 1; i <= len(ref); i++ {
|
||||
cur[0] = i
|
||||
for j := 1; j <= len(hyp); j++ {
|
||||
cost := 1
|
||||
if ref[i-1] == hyp[j-1] {
|
||||
cost = 0
|
||||
}
|
||||
cur[j] = min(prev[j]+1, min(cur[j-1]+1, prev[j-1]+cost))
|
||||
}
|
||||
prev, cur = cur, prev
|
||||
}
|
||||
return float64(prev[len(hyp)]) / float64(len(ref))
|
||||
}
|
||||
|
||||
// missingKeywords returns the keywords absent from the hypothesis. A keyword
|
||||
// matches on prefix, so a different case ending ("воды" vs "воду") does not
|
||||
// fail the assertion — the router's stage-0 grammar is stem-shaped too.
|
||||
func missingKeywords(keywords []string, hyp []string) []string {
|
||||
var missing []string
|
||||
for _, kw := range keywords {
|
||||
want := normalizeTranscript(kw)
|
||||
if len(want) == 0 {
|
||||
continue
|
||||
}
|
||||
if !containsSeq(hyp, want) {
|
||||
missing = append(missing, kw)
|
||||
}
|
||||
}
|
||||
return missing
|
||||
}
|
||||
|
||||
func containsSeq(hyp, want []string) bool {
|
||||
for i := 0; i+len(want) <= len(hyp); i++ {
|
||||
ok := true
|
||||
for j, w := range want {
|
||||
// Prefix match, so inflection differences pass but
|
||||
// distinct words do not.
|
||||
if !looseWordMatch(hyp[i+j], w) {
|
||||
ok = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if ok {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func looseWordMatch(got, want string) bool {
|
||||
if got == want {
|
||||
return true
|
||||
}
|
||||
g, w := []rune(got), []rune(want)
|
||||
n := len(w) - 1
|
||||
if len(w) > 6 {
|
||||
n = len(w) - 2
|
||||
}
|
||||
// Words of three runes or fewer have no room for a safe prefix: require
|
||||
// an exact match rather than letting "час" pass for "часть".
|
||||
if n < 3 || len(g) < n {
|
||||
return false
|
||||
}
|
||||
return string(g[:n]) == string(w[:n])
|
||||
}
|
||||
|
||||
// --- the model-backed test -------------------------------------------------
|
||||
|
||||
func TestGoldenAudioTranscription(t *testing.T) {
|
||||
m := loadGoldenManifest(t)
|
||||
|
||||
model := goldenModelPath()
|
||||
if _, err := os.Stat(model); err != nil {
|
||||
t.Skipf("whisper model %s absent (%v) — set MAVEN_WHISPER_MODEL or see AGENTS.md", model, err)
|
||||
}
|
||||
|
||||
// Same gate thresholds as mavsttd's defaults, so a regression in the
|
||||
// silence gate shows up here as an empty transcript.
|
||||
h, err := newWhisperHandler(model, 300, 0.01)
|
||||
if err != nil {
|
||||
t.Fatalf("load whisper model %s: %v", model, err)
|
||||
}
|
||||
defer h.Close()
|
||||
|
||||
for _, c := range m.Cases {
|
||||
t.Run(c.Name, func(t *testing.T) {
|
||||
path := filepath.Join("testdata", c.WAV)
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Skipf("fixture %s absent (%v) — run scripts/gen-stt-fixtures.sh", path, err)
|
||||
}
|
||||
format, pcm, err := audio.PCMFromWAV(raw)
|
||||
if err != nil {
|
||||
t.Fatalf("%s is not canonical 16k mono PCM: %v", path, err)
|
||||
}
|
||||
|
||||
resp, err := h.Transcribe(context.Background(), worker.TranscribeReq{
|
||||
Audio: audio.Audio{Format: format, Bytes: pcm},
|
||||
Lang: c.Lang,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("transcribe %s: %v", c.WAV, err)
|
||||
}
|
||||
t.Logf("%s → %q (confidence %.3f)", c.WAV, resp.Text, resp.Confidence)
|
||||
|
||||
if strings.TrimSpace(resp.Text) == "" {
|
||||
t.Fatalf("%s transcribed to empty text — the silence gate ate real speech", c.WAV)
|
||||
}
|
||||
if resp.Confidence <= 0 {
|
||||
t.Errorf("%s: confidence %v, want > 0", c.WAV, resp.Confidence)
|
||||
}
|
||||
|
||||
hyp := normalizeTranscript(resp.Text)
|
||||
ref := normalizeTranscript(c.Text)
|
||||
|
||||
if missing := missingKeywords(c.Keywords, hyp); len(missing) > 0 {
|
||||
t.Errorf("%s: missing keywords %v in %q", c.WAV, missing, resp.Text)
|
||||
}
|
||||
if wer := wordErrorRate(ref, hyp); wer > c.MaxWER {
|
||||
t.Errorf("%s: WER %.2f > %.2f\n want: %q\n got: %q", c.WAV, wer, c.MaxWER, c.Text, resp.Text)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestGoldenFixturesAreCanonical checks the committed audio without needing a
|
||||
// model, so a fixture regenerated at the wrong sample rate fails on every box.
|
||||
func TestGoldenFixturesAreCanonical(t *testing.T) {
|
||||
m := loadGoldenManifest(t)
|
||||
for _, c := range m.Cases {
|
||||
path := filepath.Join("testdata", c.WAV)
|
||||
raw, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
t.Errorf("fixture %s missing: %v", path, err)
|
||||
continue
|
||||
}
|
||||
format, pcm, err := audio.PCMFromWAV(raw)
|
||||
if err != nil {
|
||||
t.Errorf("%s: %v", path, err)
|
||||
continue
|
||||
}
|
||||
if !format.IsValid() {
|
||||
t.Errorf("%s: format %+v is not canonical", path, format)
|
||||
}
|
||||
a := audio.Audio{Format: format, Bytes: pcm}
|
||||
if d := a.Duration(); d < 0.5 || d > 10 {
|
||||
t.Errorf("%s: duration %.2fs outside the sane 0.5–10s fixture range", path, d)
|
||||
}
|
||||
// The fixture must clear mavsttd's own silence gate, otherwise the
|
||||
// model test below would be asserting on a gated empty string.
|
||||
if reason := gateReason(pcmToF32(pcm), whisperSampleRate, 300, 0.01); reason != "" {
|
||||
t.Errorf("%s: would be gated as %s", path, reason)
|
||||
}
|
||||
if len(c.Keywords) == 0 {
|
||||
t.Errorf("%s: manifest case has no keywords", c.Name)
|
||||
}
|
||||
if c.MaxWER <= 0 || c.MaxWER > 1 {
|
||||
t.Errorf("%s: max_wer %v outside (0,1]", c.Name, c.MaxWER)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func pcmToF32(b []byte) []float32 {
|
||||
out := make([]float32, len(b)/2)
|
||||
for i := range out {
|
||||
s := int16(b[i*2]) | int16(b[i*2+1])<<8
|
||||
out[i] = float32(s) / 32768.0
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// --- matcher unit tests (no model, no fixtures) ----------------------------
|
||||
|
||||
func TestNormalizeTranscript(t *testing.T) {
|
||||
got := normalizeTranscript(" Ещё, Раз... ")
|
||||
want := []string{"еще", "раз"}
|
||||
if len(got) != len(want) || got[0] != want[0] || got[1] != want[1] {
|
||||
t.Fatalf("normalizeTranscript = %v, want %v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWordErrorRate(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
ref, hyp string
|
||||
want float64
|
||||
}{
|
||||
{"identical", "напомни мне через час", "Напомни мне через час.", 0},
|
||||
{"one substitution", "напомни мне через час", "напомни мне через день", 0.25},
|
||||
{"one deletion", "напомни мне через час", "напомни мне час", 0.25},
|
||||
{"empty hypothesis", "напомни мне", "", 1},
|
||||
{"both empty", "", "", 0},
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := wordErrorRate(normalizeTranscript(c.ref), normalizeTranscript(c.hyp))
|
||||
if got != c.want {
|
||||
t.Fatalf("WER = %v, want %v", got, c.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMissingKeywords(t *testing.T) {
|
||||
hyp := normalizeTranscript("Отметь, что я выпил воду.")
|
||||
if got := missingKeywords([]string{"воды", "отметь"}, hyp); len(got) != 0 {
|
||||
t.Fatalf("missingKeywords = %v, want none (inflection must not fail the match)", got)
|
||||
}
|
||||
if got := missingKeywords([]string{"календарю"}, hyp); len(got) != 1 {
|
||||
t.Fatalf("missingKeywords = %v, want the absent keyword reported", got)
|
||||
}
|
||||
// A short word must match exactly — no 4-rune prefix shortcut that would
|
||||
// let "час" pass for "часть".
|
||||
hyp2 := normalizeTranscript("через час")
|
||||
if got := missingKeywords([]string{"часть"}, hyp2); len(got) != 1 {
|
||||
t.Fatalf("missingKeywords = %v, want %q reported missing", got, "часть")
|
||||
}
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
Vendored
+37
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"note": "Golden STT fixtures. Audio is piper-synthesised, not recorded — see scripts/gen-stt-fixtures.sh. Regenerate with that script; do not hand-edit `wav`.",
|
||||
"cases": [
|
||||
{
|
||||
"name": "ru_reminder",
|
||||
"wav": "ru_reminder.wav",
|
||||
"lang": "ru",
|
||||
"text": "напомни мне через час позвонить маме",
|
||||
"keywords": ["напомни", "час", "позвонить"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "ru_fact",
|
||||
"wav": "ru_fact.wav",
|
||||
"lang": "ru",
|
||||
"text": "отметь что я выпил воды",
|
||||
"keywords": ["отметь", "воды"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "ru_query",
|
||||
"wav": "ru_query.wav",
|
||||
"lang": "ru",
|
||||
"text": "что у меня сегодня по календарю",
|
||||
"keywords": ["сегодня", "календарю"],
|
||||
"max_wer": 0.34
|
||||
},
|
||||
{
|
||||
"name": "en_act",
|
||||
"wav": "en_act.wav",
|
||||
"lang": "en",
|
||||
"text": "restart the web server and check the disk space",
|
||||
"keywords": ["restart", "server", "disk"],
|
||||
"max_wer": 0.34
|
||||
}
|
||||
]
|
||||
}
|
||||
Vendored
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
Executable
+67
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env bash
|
||||
# gen-stt-fixtures.sh — regenerate the golden STT audio fixtures.
|
||||
#
|
||||
# The fixtures in cmd/mavsttd/testdata/*.wav are SYNTHESISED, not recorded.
|
||||
# They come out of the same piper voices maven speaks with, so nothing of the
|
||||
# owner's voice is committed and every fixture is reproducible from this
|
||||
# script plus the voice model. They are also small: 16 kHz mono s16le, a
|
||||
# couple of seconds each.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/gen-stt-fixtures.sh
|
||||
#
|
||||
# Voices are picked up from, in order, $PIPER_VOICE_RU / $PIPER_VOICE_EN, then
|
||||
# the repo's models/tts, then ~/esp-server/voices. The English voice is not
|
||||
# vendored; if it is missing the English fixture is skipped and the existing
|
||||
# one is left alone.
|
||||
set -euo pipefail
|
||||
|
||||
root="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
out="$root/cmd/mavsttd/testdata"
|
||||
piper="${PIPER_BIN:-$root/deps/piper/piper}"
|
||||
espeak="${PIPER_ESPEAK:-$root/deps/piper/espeak-ng-data}"
|
||||
|
||||
pick_voice() {
|
||||
for c in "$@"; do
|
||||
[ -f "$c" ] && { echo "$c"; return 0; }
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
ru="$(pick_voice "${PIPER_VOICE_RU:-}" "$root/models/tts/ru_RU-irina-medium.onnx" "$HOME/esp-server/voices/ru_RU-irina-medium.onnx")" || {
|
||||
echo "no russian piper voice found" >&2
|
||||
exit 1
|
||||
}
|
||||
en="$(pick_voice "${PIPER_VOICE_EN:-}" "$root/models/tts/en_US-lessac-medium.onnx" "$HOME/esp-server/voices/en_US-lessac-medium.onnx")" || en=""
|
||||
|
||||
# synth <voice> <out.wav> <text>
|
||||
# piper emits raw 22050 Hz s16le on stdout; ffmpeg resamples to the canonical
|
||||
# 16 kHz mono and writes a plain 44-byte-header WAV (-fflags bitexact keeps
|
||||
# ffmpeg's encoder LIST chunk out, so the bytes are stable across ffmpeg
|
||||
# builds and internal/audio.PCMFromWAV reads them without scanning).
|
||||
synth() {
|
||||
local voice="$1" dest="$2" text="$3"
|
||||
printf '%s' "$text" | LD_LIBRARY_PATH="$(dirname "$piper")" "$piper" \
|
||||
--model "$voice" --config "$voice.json" \
|
||||
--espeak_data "$espeak" --output_raw --quiet |
|
||||
ffmpeg -hide_banner -loglevel error -y \
|
||||
-f s16le -ar 22050 -ac 1 -i - \
|
||||
-af "adelay=200,apad=pad_dur=0.2" \
|
||||
-ar 16000 -ac 1 -c:a pcm_s16le -fflags bitexact "$dest"
|
||||
echo "wrote $dest ($(stat -c%s "$dest") bytes)"
|
||||
}
|
||||
|
||||
synth "$ru" "$out/ru_reminder.wav" "Напомни мне через час позвонить маме."
|
||||
synth "$ru" "$out/ru_fact.wav" "Отметь, что я выпил воды."
|
||||
synth "$ru" "$out/ru_query.wav" "Что у меня сегодня по календарю?"
|
||||
|
||||
if [ -n "$en" ]; then
|
||||
# Keep the English line free of words piper spells out letter by letter —
|
||||
# "nginx" comes out of lessac as "engine X", which is a TTS artefact and
|
||||
# would make the fixture assert on the wrong thing.
|
||||
synth "$en" "$out/en_act.wav" "Restart the web server and check the disk space."
|
||||
else
|
||||
echo "no english piper voice found — skipping en_act.wav" >&2
|
||||
fi
|
||||
|
||||
echo "fixtures regenerated; expected transcripts live in $out/golden_v1.json"
|
||||
Reference in New Issue
Block a user