c82dbd1e65
Both chatReq sites sent a hardcoded 0.7 and the remote path had its own const, so the one dial that governs how much a 1.7B invents could not be turned from outside the package. Config.Temperature now feeds both, 0 still means 0.7, and world.go reads the same accessor so resident and remote cannot drift. TestTalkTemperatureSweep scores the talk fixture at 0.7, 0.4, 0.2 and near greedy, three runs each so the noise band is visible. Opt-in twice (MAVEN_LLM_URL and MAVEN_TEMP_SWEEP) because it costs upwards of twenty minutes on the CPU floor. It reports and asserts nothing: the composite is not the number to read. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
146 lines
5.4 KiB
Go
146 lines
5.4 KiB
Go
package phraser
|
|
|
|
import (
|
|
"context"
|
|
"errors"
|
|
"log"
|
|
|
|
"github.com/kami/maven/internal/llm"
|
|
)
|
|
|
|
// Remote — the workstation model, seen from the phraser. `*llm.Pair` satisfies
|
|
// it, and a test fake satisfies it in three lines.
|
|
//
|
|
// Only the refusing half of Pair is here on purpose. Pair.Complete falls back to
|
|
// its own floor client, and the phraser already owns a floor: the llama-server it
|
|
// spawned. Two floors under one call is one too many, so the phraser asks whether
|
|
// the remote will take work, uses it when it will, and otherwise does exactly
|
|
// what it did before this file existed.
|
|
type Remote interface {
|
|
// Available is an atomic read of a cached probe, so it is free to call per
|
|
// turn. See llm.Pair.
|
|
Available() bool
|
|
// CompleteRemote runs on the workstation or returns ErrRemoteUnavailable. It
|
|
// never falls back.
|
|
CompleteRemote(ctx context.Context, r llm.Req) (string, error)
|
|
}
|
|
|
|
// ErrNoWorldModel — a world question was asked, a workstation model is
|
|
// configured to answer it, and that machine is not answering. The caller turns
|
|
// this into a gap he is told about ("не могу сейчас"), never into an answer from
|
|
// the resident model.
|
|
//
|
|
// This is the naming half of the degradation rule in docs/offload.md. The
|
|
// resident Qwen3-1.7B does not answer a world question worse than the 12B, it
|
|
// invents: measured, the workstation model scores knowledge 9/9 on the talk
|
|
// fixture against the resident model's confabulations
|
|
// (docs/evals/2026-08-02-workstation-gemma4-12b.md).
|
|
var ErrNoWorldModel = errors.New("phraser: no world model available")
|
|
|
|
// defaultChatTemperature — what the phraser's own transport has always sampled
|
|
// at, and what Config.Temperature falls back to. Named so the remote path
|
|
// cannot drift from the resident one silently.
|
|
const defaultChatTemperature = 0.7
|
|
|
|
// temperature — the sampling temperature for every phrasing call, resident or
|
|
// remote. Both paths read this, so a sweep moves them together.
|
|
func (p *LLMPhraser) temperature() float64 {
|
|
if p.cfg.Temperature > 0 {
|
|
return p.cfg.Temperature
|
|
}
|
|
return defaultChatTemperature
|
|
}
|
|
|
|
// UseRemote points the phraser at the workstation model. Wiring time only, once,
|
|
// before anything phrases: the field is read without a lock on every call
|
|
// because a per-turn lock to answer a question that changes at deploy time is
|
|
// not worth paying for.
|
|
//
|
|
// A nil remote is the normal state of a box with no `workstation` block, and it
|
|
// must behave exactly as the box behaved before this seam existed.
|
|
func (p *LLMPhraser) UseRemote(r Remote) {
|
|
p.remote = r
|
|
}
|
|
|
|
// PhraseWorld answers a question about the world — either from the model's own
|
|
// knowledge (no sources) or from a passage someone fetched (a live search, a ZIM
|
|
// article, a page he named). Three outcomes, and the middle one is the point:
|
|
//
|
|
// - No workstation configured. The resident model answers, exactly as it does
|
|
// today. Naming a gap needs a gap: on a box that never had a second model,
|
|
// refusing every world question would remove a capability he has now.
|
|
// - Workstation configured and taking work. It answers.
|
|
// - Workstation configured and down. ErrNoWorldModel, and the caller says so.
|
|
//
|
|
// The prompts are the ones PhraseQuery uses, built by the same two functions, so
|
|
// the two models are asked the same question in the same words.
|
|
func (p *LLMPhraser) PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error) {
|
|
sources = nonEmpty(sources)
|
|
if p.remote == nil {
|
|
return p.PhraseQuery(ctx, utterance, sources)
|
|
}
|
|
var sys, user string
|
|
if len(sources) == 0 {
|
|
sys, user = p.knowledgePrompt(utterance)
|
|
} else {
|
|
sys, user = p.evidencePrompt(utterance, sources)
|
|
}
|
|
if !p.remote.Available() {
|
|
return "", ErrNoWorldModel
|
|
}
|
|
resp, err := p.remote.CompleteRemote(ctx, llm.Req{
|
|
System: sys,
|
|
User: user,
|
|
Grammar: p.grammar(),
|
|
MaxTokens: 768,
|
|
Temperature: p.temperature(),
|
|
})
|
|
if err != nil {
|
|
// The cached probe was one interval stale, or the card went away
|
|
// mid-request. Either way this is the gap, not an error to log and
|
|
// paper over with the smaller model.
|
|
log.Printf("phraser: world model: %v", err)
|
|
return "", errors.Join(ErrNoWorldModel, err)
|
|
}
|
|
resp = stripThink(resp)
|
|
text, _, perr := parseResponseMood(resp)
|
|
if perr != nil {
|
|
log.Printf("phraser: PhraseWorld: %v", perr)
|
|
return "", errors.Join(ErrNoWorldModel, perr)
|
|
}
|
|
if text != "" {
|
|
return text, nil
|
|
}
|
|
if resp == "" {
|
|
return "", ErrNoWorldModel
|
|
}
|
|
return resp, nil
|
|
}
|
|
|
|
// remoteChat is the silent half, for the phrasing paths where the workstation
|
|
// model is only better: a nudge, a reminder, a reply, a question answered from
|
|
// his own notes. It reports whether it answered; it never reports why not,
|
|
// because the caller's next move is the resident model either way.
|
|
//
|
|
// He is not told which of the two models phrased his reply. That is the rule.
|
|
func (p *LLMPhraser) remoteChat(ctx context.Context, system, user string, maxTokens int) (string, bool) {
|
|
if p.remote == nil || !p.remote.Available() {
|
|
return "", false
|
|
}
|
|
out, err := p.remote.CompleteRemote(ctx, llm.Req{
|
|
System: system,
|
|
User: user,
|
|
Grammar: p.grammar(),
|
|
MaxTokens: maxTokens,
|
|
Temperature: p.temperature(),
|
|
})
|
|
if err != nil {
|
|
log.Printf("phraser: workstation model declined, phrasing here instead: %v", err)
|
|
return "", false
|
|
}
|
|
if out = stripThink(out); out == "" {
|
|
return "", false
|
|
}
|
|
return out, true
|
|
}
|