Files
Maven/internal/phraser/failure_test.go
claude 4534101d10 phraser: the reminder summary is cut in runes, and silence is an error (V-620)
Three defects in internal/phraser, all of the shape "reports done when
nothing happened".

The reminder summary was cut in bytes: `len(s) > 60` and `s[:57]`, in two
copies (Stub.PhraseReminder and LLMPhraser.PhraseReminder). On Russian a
letter is two bytes, so the cut fell at about 28 letters instead of 60 and
landed inside a letter about half the time. Sendable.Summary is what
voicesink hands to piper and what the telegram sink posts, so the half rune
was spoken and sent. One rune-counting helper now, shared by both. The test
that covered this was ASCII, which is what let the arithmetic stand.

The evidence branch of PhraseQuery and the bare-prose tail of PhraseChat
both returned ("", nil) when the server answered and the model wrote no
tokens. The knowledge branch has guarded that with errEmptyResponse since it
was written; these two did not. The daemon's callers check for the empty
string and paper over it, so the visible cost was the eval, which scored a
silent model as bad phrasing rather than as a failure, and a log line that
never appeared.

Stub.PhraseReminder set no Mood. Its sibling PhraseNudge sets "neutral" and
says in a comment why: the Stub is a production fallback and owes the output
contract a value. The zero value is not one of the five moods.

No prompt and no spoken wording changed, so the phrasing eval is unmoved.
2026-08-06 05:01:19 +04:00

114 lines
4.3 KiB
Go

package phraser
import (
"context"
"net/http"
"net/http/httptest"
"strings"
"testing"
)
// isFallback — the text she says is picked from that entry's variants, so a test
// pins the entry rather than the wording. Pinning one line would make editing
// fallbacks_ru_v1.json break Go tests, which is the coupling this file removed.
func isFallback(t *testing.T, key, sources, got string) bool {
t.Helper()
return DefaultFallbacks().deck().Matches(key, map[string]string{"sources": sources}, got)
}
// A dead server must be distinguishable from bad phrasing. Both PhraseChat and
// PhraseQuery keep the turn alive with canned text — and every one of those
// lines is also a legitimate reply, so the text alone cannot say which happened.
// The error is the only signal, and before Vikunja #397 it was dropped: the talk
// scorer reported a full run with zero errors off a server that answered nothing.
func TestPhrasingReportsTheFailureWithTheFallback(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
http.Error(w, "model not loaded", http.StatusServiceUnavailable)
}))
t.Cleanup(srv.Close)
p := NewLLMPhraserAt(srv.URL, Config{})
cases := []struct {
name string
call func() (string, error)
key string
sources string
}{
{"chat", func() (string, error) {
return p.PhraseChat(context.Background(), "как дела", nil)
}, fbChat, ""},
{"knowledge", func() (string, error) {
return p.PhraseQuery(context.Background(), "кто написал войну и мир", nil)
}, fbQueryUnknown, ""},
{"evidence", func() (string, error) {
return p.PhraseQuery(context.Background(), "сколько воды я выпил", []string{"два литра"})
}, fbQuerySources, "два литра"},
}
for _, c := range cases {
t.Run(c.name, func(t *testing.T) {
got, err := c.call()
if err == nil {
t.Fatalf("no error from a dead server; the scorer would count this as bad phrasing")
}
if !isFallback(t, c.key, c.sources, got) {
t.Errorf("fallback text = %q, want a %q variant — the daemon still has to say something", got, c.key)
}
})
}
}
// An empty answer is a failure too: the server is up and produced no tokens,
// which is not an answer and must not score as one.
func TestEmptyKnowledgeAnswerIsAnError(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
w.Write([]byte(`{"choices":[{"message":{"content":""}}]}`))
}))
t.Cleanup(srv.Close)
p := NewLLMPhraserAt(srv.URL, Config{})
got, err := p.PhraseQuery(context.Background(), "кто написал войну и мир", nil)
if err == nil {
t.Fatal("an empty response scored as an answer")
}
if !isFallback(t, fbQueryUnknown, "", got) {
t.Errorf("fallback text = %q, want a %q variant", got, fbQueryUnknown)
}
if !strings.Contains(err.Error(), "empty") {
t.Errorf("error = %v; want it to name the empty response", err)
}
}
// The other two paths, which had no such guard. The evidence branch of
// PhraseQuery and PhraseChat both returned ("", nil) off a server that produced
// no tokens — an empty answer reported as a successful phrasing. The daemon's
// callers check for the empty string and paper over it; the eval does not, and
// scored a silent model as bad phrasing rather than as a failure.
func TestEmptyAnswerIsAnErrorOnEveryPath(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
w.Write([]byte(`{"choices":[{"message":{"content":""}}]}`))
}))
t.Cleanup(srv.Close)
p := NewLLMPhraserAt(srv.URL, Config{})
t.Run("evidence", func(t *testing.T) {
got, err := p.PhraseQuery(context.Background(), "сколько воды я выпил", []string{"два литра"})
if err == nil {
t.Fatal("an empty response scored as an answer")
}
if !isFallback(t, fbQuerySources, "два литра", got) {
t.Errorf("fallback text = %q, want a %q variant", got, fbQuerySources)
}
})
t.Run("chat", func(t *testing.T) {
got, err := p.PhraseChat(context.Background(), "как дела", nil)
if err == nil {
t.Fatal("an empty response scored as an answer")
}
if !isFallback(t, fbChat, "", got) {
t.Errorf("fallback text = %q, want a %q variant", got, fbChat)
}
})
}