4534101d10
Three defects in internal/phraser, all of the shape "reports done when
nothing happened".
The reminder summary was cut in bytes: `len(s) > 60` and `s[:57]`, in two
copies (Stub.PhraseReminder and LLMPhraser.PhraseReminder). On Russian a
letter is two bytes, so the cut fell at about 28 letters instead of 60 and
landed inside a letter about half the time. Sendable.Summary is what
voicesink hands to piper and what the telegram sink posts, so the half rune
was spoken and sent. One rune-counting helper now, shared by both. The test
that covered this was ASCII, which is what let the arithmetic stand.
The evidence branch of PhraseQuery and the bare-prose tail of PhraseChat
both returned ("", nil) when the server answered and the model wrote no
tokens. The knowledge branch has guarded that with errEmptyResponse since it
was written; these two did not. The daemon's callers check for the empty
string and paper over it, so the visible cost was the eval, which scored a
silent model as bad phrasing rather than as a failure, and a log line that
never appeared.
Stub.PhraseReminder set no Mood. Its sibling PhraseNudge sets "neutral" and
says in a comment why: the Stub is a production fallback and owes the output
contract a value. The zero value is not one of the five moods.
No prompt and no spoken wording changed, so the phrasing eval is unmoved.
114 lines
4.3 KiB
Go
114 lines
4.3 KiB
Go
package phraser
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"strings"
|
|
"testing"
|
|
)
|
|
|
|
// isFallback — the text she says is picked from that entry's variants, so a test
|
|
// pins the entry rather than the wording. Pinning one line would make editing
|
|
// fallbacks_ru_v1.json break Go tests, which is the coupling this file removed.
|
|
func isFallback(t *testing.T, key, sources, got string) bool {
|
|
t.Helper()
|
|
return DefaultFallbacks().deck().Matches(key, map[string]string{"sources": sources}, got)
|
|
}
|
|
|
|
// A dead server must be distinguishable from bad phrasing. Both PhraseChat and
|
|
// PhraseQuery keep the turn alive with canned text — and every one of those
|
|
// lines is also a legitimate reply, so the text alone cannot say which happened.
|
|
// The error is the only signal, and before Vikunja #397 it was dropped: the talk
|
|
// scorer reported a full run with zero errors off a server that answered nothing.
|
|
func TestPhrasingReportsTheFailureWithTheFallback(t *testing.T) {
|
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
http.Error(w, "model not loaded", http.StatusServiceUnavailable)
|
|
}))
|
|
t.Cleanup(srv.Close)
|
|
p := NewLLMPhraserAt(srv.URL, Config{})
|
|
|
|
cases := []struct {
|
|
name string
|
|
call func() (string, error)
|
|
key string
|
|
sources string
|
|
}{
|
|
{"chat", func() (string, error) {
|
|
return p.PhraseChat(context.Background(), "как дела", nil)
|
|
}, fbChat, ""},
|
|
{"knowledge", func() (string, error) {
|
|
return p.PhraseQuery(context.Background(), "кто написал войну и мир", nil)
|
|
}, fbQueryUnknown, ""},
|
|
{"evidence", func() (string, error) {
|
|
return p.PhraseQuery(context.Background(), "сколько воды я выпил", []string{"два литра"})
|
|
}, fbQuerySources, "два литра"},
|
|
}
|
|
for _, c := range cases {
|
|
t.Run(c.name, func(t *testing.T) {
|
|
got, err := c.call()
|
|
if err == nil {
|
|
t.Fatalf("no error from a dead server; the scorer would count this as bad phrasing")
|
|
}
|
|
if !isFallback(t, c.key, c.sources, got) {
|
|
t.Errorf("fallback text = %q, want a %q variant — the daemon still has to say something", got, c.key)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// An empty answer is a failure too: the server is up and produced no tokens,
|
|
// which is not an answer and must not score as one.
|
|
func TestEmptyKnowledgeAnswerIsAnError(t *testing.T) {
|
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
w.Write([]byte(`{"choices":[{"message":{"content":""}}]}`))
|
|
}))
|
|
t.Cleanup(srv.Close)
|
|
p := NewLLMPhraserAt(srv.URL, Config{})
|
|
|
|
got, err := p.PhraseQuery(context.Background(), "кто написал войну и мир", nil)
|
|
if err == nil {
|
|
t.Fatal("an empty response scored as an answer")
|
|
}
|
|
if !isFallback(t, fbQueryUnknown, "", got) {
|
|
t.Errorf("fallback text = %q, want a %q variant", got, fbQueryUnknown)
|
|
}
|
|
if !strings.Contains(err.Error(), "empty") {
|
|
t.Errorf("error = %v; want it to name the empty response", err)
|
|
}
|
|
}
|
|
|
|
// The other two paths, which had no such guard. The evidence branch of
|
|
// PhraseQuery and PhraseChat both returned ("", nil) off a server that produced
|
|
// no tokens — an empty answer reported as a successful phrasing. The daemon's
|
|
// callers check for the empty string and paper over it; the eval does not, and
|
|
// scored a silent model as bad phrasing rather than as a failure.
|
|
func TestEmptyAnswerIsAnErrorOnEveryPath(t *testing.T) {
|
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
w.Header().Set("Content-Type", "application/json")
|
|
w.Write([]byte(`{"choices":[{"message":{"content":""}}]}`))
|
|
}))
|
|
t.Cleanup(srv.Close)
|
|
p := NewLLMPhraserAt(srv.URL, Config{})
|
|
|
|
t.Run("evidence", func(t *testing.T) {
|
|
got, err := p.PhraseQuery(context.Background(), "сколько воды я выпил", []string{"два литра"})
|
|
if err == nil {
|
|
t.Fatal("an empty response scored as an answer")
|
|
}
|
|
if !isFallback(t, fbQuerySources, "два литра", got) {
|
|
t.Errorf("fallback text = %q, want a %q variant", got, fbQuerySources)
|
|
}
|
|
})
|
|
t.Run("chat", func(t *testing.T) {
|
|
got, err := p.PhraseChat(context.Background(), "как дела", nil)
|
|
if err == nil {
|
|
t.Fatal("an empty response scored as an answer")
|
|
}
|
|
if !isFallback(t, fbChat, "", got) {
|
|
t.Errorf("fallback text = %q, want a %q variant", got, fbChat)
|
|
}
|
|
})
|
|
}
|