Merge branch 'fix/g07' into fix/integrated
# Conflicts: # internal/ipc/api.go # internal/ipc/client.go # internal/llm/client.go
This commit is contained in:
@@ -58,6 +58,7 @@ func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
|
||||
ModelPath: res.ModelPath,
|
||||
BaseURL: res.BaseURL,
|
||||
RolledBack: res.RolledBack,
|
||||
NoBackend: res.NoBackend,
|
||||
TookMs: res.Took.Milliseconds(),
|
||||
}
|
||||
if err != nil {
|
||||
@@ -106,9 +107,15 @@ func wireModelSwap(srv *ipc.Server, phr phraser.Phraser, cfg *config.Config) {
|
||||
// the port of a server that no longer exists, and the daemon would degrade to
|
||||
// the classifier permanently after the first swap. The client is re-pointed, not
|
||||
// rebuilt, so nothing that holds it has to know a swap happened.
|
||||
// SetSwapGate is the other half, and on the deploy shape it is the load-bearing
|
||||
// one:
|
||||
// llama-server is relaunched on the same fixed port, so SetBaseURL is usually a
|
||||
// no-op, while the gate is what makes the swap's drain count these callers at
|
||||
// all. Without it a swap can kill the server mid-routing-decision.
|
||||
func llmClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
||||
c := llm.New(lp.BaseURL(), timeout)
|
||||
c.SetGate(residentGate, false)
|
||||
c.SetSwapGate(lp)
|
||||
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
||||
return c
|
||||
}
|
||||
@@ -136,6 +143,7 @@ var residentGate = llm.NewGate(backgroundQuiet)
|
||||
func llmBackgroundClientFor(lp *phraser.LLMPhraser, timeout time.Duration) *llm.Client {
|
||||
c := llm.New(lp.BaseURL(), timeout)
|
||||
c.SetGate(residentGate, true)
|
||||
c.SetSwapGate(lp)
|
||||
lp.OnSwap(func(base string) { c.SetBaseURL(base) })
|
||||
return c
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user