Merge task/467 so the sweep tail can reach the three mechanisms (V-528)
attentionq.go, repair.go and internal/router/complaint.go carry the last
hand-written Russian patterns of the V-522 sweep, and they live on task/467.
internal/lexicon, internal/morph and cmd/mavend/topics.go live here. One of
the two had to move.
Four conflicts, and one of them is a real collision rather than a mechanical
one. Both branches wrote the narrative stage 0 rule. This side had
NarrativeQueryGrammars, plural, with the rest-of-day rule beside it and the
verb alternation built from the lexicon; task/467 had NarrativeQueryGrammar,
singular, which extracts the topic into Slots.Text, refuses a bare "расскажи",
and excludes the shapes that are chat ("расскажи о себе", "историю на ночь").
Resolved by keeping this side's container and this side's lexicon-built
pattern, and taking every behaviour only the other side had: the topic slot,
the empty-topic refusal, chatNarrativeTopics, and its wiring position after
TaskCaptureGrammar so "запиши" still beats "расскажи".
The rest: queryFeeds keeps task/467's conditional claim (V-474 supersedes the
unconditional one), rank.go keeps Spoken and drops pluralTasksRU because
say.CountWord is the one copy of Russian count agreement, and vendor/ was
re-vendored — the merged modules.txt claimed replaces for nexus and praxis
that neither go.mod has.
Routing fixture 58/82, unchanged from both sides.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -91,6 +91,15 @@ type Config struct {
|
||||
NCtx int
|
||||
Timeout time.Duration
|
||||
|
||||
// StartupTimeout bounds the wait for llama-server to print the address it
|
||||
// listens on. A config field and not a constant because the box may
|
||||
// legitimately need longer: a cold 1.7B loading off a spinning disk can
|
||||
// outrun a minute, and until this existed that returned "server did not
|
||||
// start within 60s" with no way to raise it.
|
||||
//
|
||||
// 0 ⇒ defaultStartupTimeout.
|
||||
StartupTimeout time.Duration
|
||||
|
||||
// CacheRAMMiB bounds llama-server's prompt cache, which is what actually ate
|
||||
// this box. Measured on homesrv 2026-08-03: the server's own default limit is
|
||||
// 8192 MiB, it stores the full KV state of every idle slot it evicts (112 kiB
|
||||
@@ -140,6 +149,8 @@ func DefaultConfig(modelPath string) Config {
|
||||
// 512 MiB caps total RSS near 1 GB and still holds several recent prompts.
|
||||
CacheRAMMiB: 512,
|
||||
Timeout: 30 * time.Second,
|
||||
|
||||
StartupTimeout: defaultStartupTimeout,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -266,7 +277,15 @@ func llamaArgs(cfg Config) []string {
|
||||
return args
|
||||
}
|
||||
|
||||
// defaultStartupTimeout — the wait for llama-server's listen line when Config
|
||||
// does not set one. A cold model load off disk is the slow part.
|
||||
const defaultStartupTimeout = 60 * time.Second
|
||||
|
||||
func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
startupTimeout := cfg.StartupTimeout
|
||||
if startupTimeout <= 0 {
|
||||
startupTimeout = defaultStartupTimeout
|
||||
}
|
||||
p := &llamaProc{}
|
||||
cmd := exec.CommandContext(ctx, cfg.BinPath, llamaArgs(cfg)...)
|
||||
// Pdeathsig: the kernel SIGKILLs llama-server the moment mavend dies — by
|
||||
@@ -344,8 +363,8 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
||||
return fail(fmt.Errorf("llm: server output: %w; last output: %s", err, tail.String()))
|
||||
case <-ctx.Done():
|
||||
return fail(ctx.Err())
|
||||
case <-time.After(60 * time.Second):
|
||||
return fail(fmt.Errorf("llm: server did not start within 60s; last output: %s", tail.String()))
|
||||
case <-time.After(startupTimeout):
|
||||
return fail(fmt.Errorf("llm: server did not start within %s; last output: %s", startupTimeout, tail.String()))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -181,6 +181,37 @@ exit 1`)
|
||||
t.Fatalf("err = %v, want context.Canceled", err)
|
||||
}
|
||||
})
|
||||
|
||||
// The last arm of the startup race, and the one most likely to leak: a
|
||||
// llama-server still loading a model is alive, so giving up on it without
|
||||
// killing and reaping it orphans a process holding the GPU. Testable at all
|
||||
// because Config.StartupTimeout replaced a hardcoded 60s (Vikunja #323).
|
||||
t.Run("startup timeout", func(t *testing.T) {
|
||||
pidPath := filepath.Join(t.TempDir(), "pid")
|
||||
bin := fakeLlama(t, fmt.Sprintf(`echo $$ > %s
|
||||
while : ; do sleep 1 ; done`, pidPath))
|
||||
cfg := testCfg(bin)
|
||||
cfg.StartupTimeout = 200 * time.Millisecond
|
||||
|
||||
_, err := startLlamaProc(context.Background(), cfg)
|
||||
if err == nil || !strings.Contains(err.Error(), "did not start within 200ms") {
|
||||
t.Fatalf("err = %v, want the startup-timeout arm naming the timeout", err)
|
||||
}
|
||||
|
||||
raw, readErr := os.ReadFile(pidPath)
|
||||
if readErr != nil {
|
||||
t.Fatalf("fake server never recorded its pid: %v", readErr)
|
||||
}
|
||||
pid, convErr := strconv.Atoi(strings.TrimSpace(string(raw)))
|
||||
if convErr != nil {
|
||||
t.Fatalf("pid file = %q: %v", raw, convErr)
|
||||
}
|
||||
// Killed, and reaped: a zombie still answers signal 0, so this asserts
|
||||
// the Wait ran too.
|
||||
if err := syscall.Kill(pid, 0); err == nil {
|
||||
t.Errorf("llama-server %d survived the startup timeout", pid)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestNewLLMPhraserSpawns(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user