From f9b2391a8b5374ad67dc05bdd1b368ff4e9ee48d Mon Sep 17 00:00:00 2001 From: claude Date: Mon, 3 Aug 2026 23:19:15 +0400 Subject: [PATCH] phraser: cap llama-server's prompt cache at 512 MiB (V-499) The forwarded log named the cause in one line: the prompt cache limit defaults to 8192 MiB. llama-server saves the full KV state of every idle slot it evicts, 112 kiB per token, so RSS climbed about 170MB per distinct prompt until the deployed server held 7.9GB for a 1.1GB model. Measured on homesrv today, uncapped versus `--cache-ram 512`: RSS plateaus at 932MB from the fourth distinct prompt instead of climbing. The task's leading guess was wrong. `-ngl 99` costs almost no RSS, because RADV keeps device memory outside the process. Numbers and method in docs/evals/2026-08-03-llama-prompt-cache.md. `-c 4096` is untouched. The knob is `phraser.cache_ram_mib`, unset means 512, negative passes no flag for a llama-server too old to know it. The deploy still runs the old image, so the box keeps its 8 GiB default until mavend is rebuilt. Co-Authored-By: Claude Opus 5 --- cmd/mavend/main.go | 18 +++++ deploy/mavend.json | 1 + docs/evals/2026-08-03-llama-prompt-cache.md | 77 +++++++++++++++++++++ internal/config/config.go | 6 ++ internal/phraser/llmphraser.go | 2 +- 5 files changed, 103 insertions(+), 1 deletion(-) create mode 100644 docs/evals/2026-08-03-llama-prompt-cache.md diff --git a/cmd/mavend/main.go b/cmd/mavend/main.go index 892330d..05d23f8 100644 --- a/cmd/mavend/main.go +++ b/cmd/mavend/main.go @@ -251,6 +251,7 @@ func run(args []string) error { Listen: cfg.Phraser.Listen, NGpuLayers: cfg.Phraser.NGpuLayers, NCtx: cfg.Phraser.NCtx, + CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB), Timeout: time.Duration(cfg.Phraser.Timeout), LLMNudges: cfg.Phraser.LLMNudges, ContextBlock: contextBlockFn(cfg, time.Now), @@ -525,6 +526,7 @@ func run(args []string) error { Listen: cfg.Phraser.Listen, NGpuLayers: cfg.Phraser.NGpuLayers, NCtx: cfg.Phraser.NCtx, + CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB), Timeout: time.Duration(cfg.Phraser.Timeout), LLMNudges: cfg.Phraser.LLMNudges, ContextBlock: contextBlockFn(cfg, time.Now), @@ -788,6 +790,22 @@ func personaFacts(cfg *config.Config) persona.Facts { return f } +// cacheRAMMiB resolves phraser.cache_ram_mib into the phraser's field. Unset +// means 512 MiB and not "whatever the server does", because the server's own +// default is 8 GiB of prompt cache and that is what put 7.9 GB of RSS and half +// a gigabyte of swap on homesrv for a 1.1 GB model. A negative value is the +// deliberate opt-out: no flag is passed, the server's default applies, and the +// operator owns the consequence. +func cacheRAMMiB(configured int) int { + if configured == 0 { + return 512 + } + if configured < 0 { + return 0 + } + return configured +} + // contextBlockFn returns the per-turn renderer of the shared context block. // Per turn, not once at startup, because the block states the current time. func contextBlockFn(cfg *config.Config, now func() time.Time) func() string { diff --git a/deploy/mavend.json b/deploy/mavend.json index 3a3ee02..bb67b34 100644 --- a/deploy/mavend.json +++ b/deploy/mavend.json @@ -20,6 +20,7 @@ "bin_path": "llama-server", "n_gpu_layers": 99, "n_ctx": 4096, + "cache_ram_mib": 512, "timeout": "60s", "llm_nudges": false }, diff --git a/docs/evals/2026-08-03-llama-prompt-cache.md b/docs/evals/2026-08-03-llama-prompt-cache.md new file mode 100644 index 0000000..1a7c48d --- /dev/null +++ b/docs/evals/2026-08-03-llama-prompt-cache.md @@ -0,0 +1,77 @@ +# Where the resident model's 7.9GB of RSS goes (2026-08-03, homesrv) + +Measured for Vikunja #499. The deployed llama-server held 7.9GB RSS for a 1.1GB +model file. Half a gigabyte of it was in swap, on a box that also runs +whisper.cpp, piper and the embedder. + +## Method + +`maven-mavend-1` was stopped for the measurement, with the owner's approval. +Its own binary then ran on the host with the exact deployed command line. That +binary is `/opt/maven/bin/llama-server`, version `1 (4c65955)`, a Vulkan build. + +```sh +llama-server -m /mnt/hdd1/llms/qwen3/Qwen3-1.7B-UD-Q4_K_XL.gguf \ + --host 127.0.0.1 --port 18099 -c 4096 -ngl 99 --no-webui +``` + +RSS was read from `/proc//status` after load and after each of 8 distinct +1521-token prompts. `smaps` of the deployed process was read first, from inside +the container, since the host user cannot read another user's maps. + +## The cause: the prompt cache, not the weights and not the offload + +The startup log says it outright: + +```text +srv load_model: prompt cache is enabled, size limit: 8192 MiB +srv llama_server: n_parallel is set to auto, using n_parallel = 4 and kv_unified = true +``` + +The server saves the full KV state of every idle slot it evicts. It keeps up to +8GiB of those states in host RAM (llama.cpp PR 16391). One saved prompt of 1521 +tokens costs 166.377 MiB. That is 112 kiB per token, exactly Qwen3-1.7B's KV +footprint (28 layers x 2 x 1024 dims x 2 bytes). + +RSS at rest, and per distinct prompt: + +| Prompts served | RSS, default | RSS, `--cache-ram 512` | +|---|---|---| +| 0 (just loaded) | 443 MB | 411 MB | +| 1 | 445 MB | 411 MB | +| 4 | 958 MB | 929 MB | +| 8 | 1641 MB | 932 MB | + +Uncapped, RSS climbs about 170MB per distinct prompt and does not stop until +the 8GiB limit. Capped at 512 MiB it plateaus at 932MB from the fourth prompt +on, with the cache holding steady at `3 prompts, 499.132 MiB` and evicting. + +The 7.9GB on the running daemon was that climb, weeks of it. Its `smaps` showed +one 6.03GB anonymous mapping at 5.32GB resident plus a 1.45GB mapping at 1.27GB +resident, and only 30MB of file-backed RSS. + +## The task's leading guess was wrong + +`-ngl 99` on the Vega iGPU costs almost no process RSS. A freshly loaded server +has 95MB of anonymous RSS in total. RADV allocates device memory through the +kernel, outside the process, and the log sees 8202 MiB free on `Vulkan0`. The +weights are mmapped and file-backed, so they are evictable and do not pin RSS. The logit buffer is not visible in the numbers above at all. + +## Decision + +`--cache-ram 512` is now the default, wired as `phraser.cache_ram_mib` and set +in `deploy/mavend.json`. 512 MiB caps total RSS near 1GB, an eighth of what the +box carried. It still holds three of the 1521-token probes above. Maven's real +routing and phrasing prompts are much shorter, so it holds more of those than +the table suggests. `-c 4096` is untouched, as #499 +required. A negative `cache_ram_mib` passes no flag, for a llama-server too old +to know it. + +Not changed: `n_parallel = 4`. With `kv_unified = true` the four slots share one +4096-token KV cache, so they do not multiply it. + +The other half of #499 was that none of these lines were reachable. mavend +scraped llama-server's stderr for the listen line and discarded it, and never +piped stdout at all. Both streams now go to mavend's log with a `llama:` prefix. +The last 12 startup lines go into the error when the server dies before it +listens. diff --git a/internal/config/config.go b/internal/config/config.go index c34d8c2..ff72b31 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -1276,6 +1276,12 @@ type PhraserConfig struct { NCtx int `json:"n_ctx,omitempty"` Timeout Duration `json:"timeout,omitempty"` + // CacheRAMMiB bounds llama-server's prompt cache. Omitted ⇒ 512 MiB, which + // is what keeps the resident model near 1 GB of RSS instead of the 7.9 GB + // measured on 2026-08-03. Set it to -1 to pass no flag at all and let the + // server apply its own 8 GiB default. See phraser.Config.CacheRAMMiB. + CacheRAMMiB int `json:"cache_ram_mib,omitempty"` + // LLMNudges — let the model word nudges again. Off by default: nudges are // worded from hand-written Russian templates now (the model broke the // persona and invented units). Chat, query and reminder phrasing always go diff --git a/internal/phraser/llmphraser.go b/internal/phraser/llmphraser.go index d0e79bf..2c5e28d 100644 --- a/internal/phraser/llmphraser.go +++ b/internal/phraser/llmphraser.go @@ -133,7 +133,7 @@ func DefaultConfig(modelPath string) Config { NCtx: 2048, // 512 MiB caps total RSS near 1 GB and still holds several recent prompts. CacheRAMMiB: 512, - Timeout: 30 * time.Second, + Timeout: 30 * time.Second, } }