9a70f7378b
It sat in world.go, which is about the workstation model; it is a phrasing error and belongs in llmphraser.go. Also trims the PhraseQuery doc. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
138 lines
5.1 KiB
Go
138 lines
5.1 KiB
Go
package phraser
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"log"
|
|
|
|
"github.com/kami/maven/internal/llm"
|
|
)
|
|
|
|
// Remote — the workstation model, seen from the phraser. `*llm.Pair` satisfies
|
|
// it, and a test fake satisfies it in three lines.
|
|
//
|
|
// Only the refusing half of Pair is here on purpose. Pair.Complete falls back to
|
|
// its own floor client, and the phraser already owns a floor: the llama-server it
|
|
// spawned. Two floors under one call is one too many, so the phraser asks whether
|
|
// the remote will take work, uses it when it will, and otherwise does exactly
|
|
// what it did before this file existed.
|
|
type Remote interface {
|
|
// Available is an atomic read of a cached probe, so it is free to call per
|
|
// turn. See llm.Pair.
|
|
Available() bool
|
|
// CompleteRemote runs on the workstation or returns ErrRemoteUnavailable. It
|
|
// never falls back.
|
|
CompleteRemote(ctx context.Context, r llm.Req) (string, error)
|
|
}
|
|
|
|
// ErrNoWorldModel — a world question was asked, a workstation model is
|
|
// configured to answer it, and that machine is not answering. The caller turns
|
|
// this into a gap he is told about ("не могу сейчас"), never into an answer from
|
|
// the resident model.
|
|
//
|
|
// This is the naming half of the degradation rule in docs/offload.md. The
|
|
// resident Qwen3-1.7B does not answer a world question worse than the 12B, it
|
|
// invents: measured, the workstation model scores knowledge 9/9 on the talk
|
|
// fixture against the resident model's confabulations
|
|
// (docs/evals/2026-08-02-workstation-gemma4-12b.md).
|
|
var ErrNoWorldModel = errors.New("phraser: no world model available")
|
|
|
|
// chatTemperature — what the phraser's own transport has always sampled at.
|
|
// Named so the remote path cannot drift from it silently. Whether 0.7 is right
|
|
// at all is Vikunja #402, and answering that here would hide a phrasing change
|
|
// inside a routing change.
|
|
const chatTemperature = 0.7
|
|
|
|
// UseRemote points the phraser at the workstation model. Wiring time only, once,
|
|
// before anything phrases: the field is read without a lock on every call
|
|
// because a per-turn lock to answer a question that changes at deploy time is
|
|
// not worth paying for.
|
|
//
|
|
// A nil remote is the normal state of a box with no `workstation` block, and it
|
|
// must behave exactly as the box behaved before this seam existed.
|
|
func (p *LLMPhraser) UseRemote(r Remote) {
|
|
p.remote = r
|
|
}
|
|
|
|
// PhraseWorld answers a question about the world — either from the model's own
|
|
// knowledge (no sources) or from a passage someone fetched (a live search, a ZIM
|
|
// article, a page he named). Three outcomes, and the middle one is the point:
|
|
//
|
|
// - No workstation configured. The resident model answers, exactly as it does
|
|
// today. Naming a gap needs a gap: on a box that never had a second model,
|
|
// refusing every world question would remove a capability he has now.
|
|
// - Workstation configured and taking work. It answers.
|
|
// - Workstation configured and down. ErrNoWorldModel, and the caller says so.
|
|
//
|
|
// The prompts are the ones PhraseQuery uses, built by the same two functions, so
|
|
// the two models are asked the same question in the same words.
|
|
func (p *LLMPhraser) PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
|
|
sources = nonEmpty(sources)
|
|
if p.remote == nil {
|
|
return p.PhraseQuery(ctx, utterance, sources)
|
|
}
|
|
var sys, user string
|
|
if len(sources) == 0 {
|
|
sys, user = p.knowledgePrompt(utterance)
|
|
} else {
|
|
sys, user = p.evidencePrompt(utterance, sources)
|
|
}
|
|
if !p.remote.Available() {
|
|
return "", ErrNoWorldModel
|
|
}
|
|
resp, err := p.remote.CompleteRemote(ctx, llm.Req{
|
|
System: sys,
|
|
User: user,
|
|
Grammar: p.grammar(),
|
|
MaxTokens: 768,
|
|
Temperature: chatTemperature,
|
|
})
|
|
if err != nil {
|
|
// The cached probe was one interval stale, or the card went away
|
|
// mid-request. Either way this is the gap, not an error to log and
|
|
// paper over with the smaller model.
|
|
log.Printf("phraser: world model: %v", err)
|
|
return "", errors.Join(ErrNoWorldModel, err)
|
|
}
|
|
resp = stripThink(resp)
|
|
text, _, perr := parseResponseMood(resp)
|
|
if perr != nil {
|
|
log.Printf("phraser: PhraseWorld: %v", perr)
|
|
return "", errors.Join(ErrNoWorldModel, perr)
|
|
}
|
|
if text != "" {
|
|
return text, nil
|
|
}
|
|
if resp == "" {
|
|
return "", ErrNoWorldModel
|
|
}
|
|
return resp, nil
|
|
}
|
|
|
|
// remoteChat is the silent half, for the phrasing paths where the workstation
|
|
// model is only better: a nudge, a reminder, a reply, a question answered from
|
|
// his own notes. It reports whether it answered; it never reports why not,
|
|
// because the caller's next move is the resident model either way.
|
|
//
|
|
// He is not told which of the two models phrased his reply. That is the rule.
|
|
func (p *LLMPhraser) remoteChat(ctx context.Context, system, user string, maxTokens int) (string, bool) {
|
|
if p.remote == nil || !p.remote.Available() {
|
|
return "", false
|
|
}
|
|
out, err := p.remote.CompleteRemote(ctx, llm.Req{
|
|
System: system,
|
|
User: user,
|
|
Grammar: p.grammar(),
|
|
MaxTokens: maxTokens,
|
|
Temperature: chatTemperature,
|
|
})
|
|
if err != nil {
|
|
log.Printf("phraser: workstation model declined, phrasing here instead: %v", err)
|
|
return "", false
|
|
}
|
|
if out = stripThink(out); out == "" {
|
|
return "", false
|
|
}
|
|
return out, true
|
|
}
|