package phraser import ( "context" "errors" "log" "github.com/kami/maven/internal/llm" ) // Remote — the workstation model, seen from the phraser. `*llm.Pair` satisfies // it, and a test fake satisfies it in three lines. // // Only the refusing half of Pair is here on purpose. Pair.Complete falls back to // its own floor client, and the phraser already owns a floor: the llama-server it // spawned. Two floors under one call is one too many, so the phraser asks whether // the remote will take work, uses it when it will, and otherwise does exactly // what it did before this file existed. type Remote interface { // Available is an atomic read of a cached probe, so it is free to call per // turn. See llm.Pair. Available() bool // CompleteRemote runs on the workstation or returns ErrRemoteUnavailable. It // never falls back. CompleteRemote(ctx context.Context, r llm.Req) (string, error) } // ErrNoWorldModel — a world question was asked, a workstation model is // configured to answer it, and that machine is not answering. The caller turns // this into a gap he is told about ("не могу сейчас"), never into an answer from // the resident model. // // This is the naming half of the degradation rule in docs/offload.md. The // resident Qwen3-1.7B does not answer a world question worse than the 12B, it // invents: measured, the workstation model scores knowledge 9/9 on the talk // fixture against the resident model's confabulations // (docs/evals/2026-08-02-workstation-gemma4-12b.md). var ErrNoWorldModel = errors.New("phraser: no world model available") // defaultChatTemperature — what the phraser's own transport has always sampled // at, and what Config.Temperature falls back to. Named so the remote path // cannot drift from the resident one silently. const defaultChatTemperature = 0.7 // temperature — the sampling temperature for every phrasing call, resident or // remote. Both paths read this, so a sweep moves them together. func (p *LLMPhraser) temperature() float64 { if p.cfg.Temperature > 0 { return p.cfg.Temperature } return defaultChatTemperature } // UseRemote points the phraser at the workstation model. Wiring time only, once, // before anything phrases: the field is read without a lock on every call // because a per-turn lock to answer a question that changes at deploy time is // not worth paying for. // // A nil remote is the normal state of a box with no `workstation` block, and it // must behave exactly as the box behaved before this seam existed. func (p *LLMPhraser) UseRemote(r Remote) { p.remote = r } // PhraseWorld answers a question about the world — either from the model's own // knowledge (no sources) or from a passage someone fetched (a live search, a ZIM // article, a page he named). Three outcomes, and the middle one is the point: // // - No workstation configured. The resident model answers, exactly as it does // today. Naming a gap needs a gap: on a box that never had a second model, // refusing every world question would remove a capability he has now. // - Workstation configured and taking work. It answers. // - Workstation configured and down. ErrNoWorldModel, and the caller says so. // // The prompts are the ones PhraseQuery uses, built by the same two functions, so // the two models are asked the same question in the same words. func (p *LLMPhraser) PhraseWorld(ctx context.Context, utterance string, sources []string) (string, error) { sources = nonEmpty(sources) if p.remote == nil { return p.PhraseQuery(ctx, utterance, sources) } var sys, user string if len(sources) == 0 { sys, user = p.knowledgePrompt(utterance) } else { sys, user = p.evidencePrompt(utterance, sources) } if !p.remote.Available() { return "", ErrNoWorldModel } resp, err := p.remote.CompleteRemote(ctx, llm.Req{ System: sys, User: user, Grammar: p.grammar(), MaxTokens: 768, Temperature: p.temperature(), }) if err != nil { // The cached probe was one interval stale, or the card went away // mid-request. Either way this is the gap, not an error to log and // paper over with the smaller model. log.Printf("phraser: world model: %v", err) return "", errors.Join(ErrNoWorldModel, err) } resp = stripThink(resp) text, _, perr := parseResponseMood(resp) if perr != nil { log.Printf("phraser: PhraseWorld: %v", perr) return "", errors.Join(ErrNoWorldModel, perr) } if text != "" { return text, nil } if resp == "" { return "", ErrNoWorldModel } return resp, nil } // remoteChat is the silent half, for the phrasing paths where the workstation // model is only better: a nudge, a reminder, a reply, a question answered from // his own notes. It reports whether it answered; it never reports why not, // because the caller's next move is the resident model either way. // // He is not told which of the two models phrased his reply. That is the rule. func (p *LLMPhraser) remoteChat(ctx context.Context, system, user string, maxTokens int) (string, bool) { if p.remote == nil || !p.remote.Available() { return "", false } out, err := p.remote.CompleteRemote(ctx, llm.Req{ System: system, User: user, Grammar: p.grammar(), MaxTokens: maxTokens, Temperature: p.temperature(), }) if err != nil { log.Printf("phraser: workstation model declined, phrasing here instead: %v", err) return "", false } if out = stripThink(out); out == "" { return "", false } return out, true }