diff --git a/cmd/mavend/main.go b/cmd/mavend/main.go index 892330d..05d23f8 100644 --- a/cmd/mavend/main.go +++ b/cmd/mavend/main.go @@ -251,6 +251,7 @@ func run(args []string) error { Listen: cfg.Phraser.Listen, NGpuLayers: cfg.Phraser.NGpuLayers, NCtx: cfg.Phraser.NCtx, + CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB), Timeout: time.Duration(cfg.Phraser.Timeout), LLMNudges: cfg.Phraser.LLMNudges, ContextBlock: contextBlockFn(cfg, time.Now), @@ -525,6 +526,7 @@ func run(args []string) error { Listen: cfg.Phraser.Listen, NGpuLayers: cfg.Phraser.NGpuLayers, NCtx: cfg.Phraser.NCtx, + CacheRAMMiB: cacheRAMMiB(cfg.Phraser.CacheRAMMiB), Timeout: time.Duration(cfg.Phraser.Timeout), LLMNudges: cfg.Phraser.LLMNudges, ContextBlock: contextBlockFn(cfg, time.Now), @@ -788,6 +790,22 @@ func personaFacts(cfg *config.Config) persona.Facts { return f } +// cacheRAMMiB resolves phraser.cache_ram_mib into the phraser's field. Unset +// means 512 MiB and not "whatever the server does", because the server's own +// default is 8 GiB of prompt cache and that is what put 7.9 GB of RSS and half +// a gigabyte of swap on homesrv for a 1.1 GB model. A negative value is the +// deliberate opt-out: no flag is passed, the server's default applies, and the +// operator owns the consequence. +func cacheRAMMiB(configured int) int { + if configured == 0 { + return 512 + } + if configured < 0 { + return 0 + } + return configured +} + // contextBlockFn returns the per-turn renderer of the shared context block. // Per turn, not once at startup, because the block states the current time. func contextBlockFn(cfg *config.Config, now func() time.Time) func() string { diff --git a/deploy/mavend.json b/deploy/mavend.json index 3a3ee02..bb67b34 100644 --- a/deploy/mavend.json +++ b/deploy/mavend.json @@ -20,6 +20,7 @@ "bin_path": "llama-server", "n_gpu_layers": 99, "n_ctx": 4096, + "cache_ram_mib": 512, "timeout": "60s", "llm_nudges": false }, diff --git a/docs/evals/2026-08-03-llama-prompt-cache.md b/docs/evals/2026-08-03-llama-prompt-cache.md new file mode 100644 index 0000000..1a7c48d --- /dev/null +++ b/docs/evals/2026-08-03-llama-prompt-cache.md @@ -0,0 +1,77 @@ +# Where the resident model's 7.9GB of RSS goes (2026-08-03, homesrv) + +Measured for Vikunja #499. The deployed llama-server held 7.9GB RSS for a 1.1GB +model file. Half a gigabyte of it was in swap, on a box that also runs +whisper.cpp, piper and the embedder. + +## Method + +`maven-mavend-1` was stopped for the measurement, with the owner's approval. +Its own binary then ran on the host with the exact deployed command line. That +binary is `/opt/maven/bin/llama-server`, version `1 (4c65955)`, a Vulkan build. + +```sh +llama-server -m /mnt/hdd1/llms/qwen3/Qwen3-1.7B-UD-Q4_K_XL.gguf \ + --host 127.0.0.1 --port 18099 -c 4096 -ngl 99 --no-webui +``` + +RSS was read from `/proc//status` after load and after each of 8 distinct +1521-token prompts. `smaps` of the deployed process was read first, from inside +the container, since the host user cannot read another user's maps. + +## The cause: the prompt cache, not the weights and not the offload + +The startup log says it outright: + +```text +srv load_model: prompt cache is enabled, size limit: 8192 MiB +srv llama_server: n_parallel is set to auto, using n_parallel = 4 and kv_unified = true +``` + +The server saves the full KV state of every idle slot it evicts. It keeps up to +8GiB of those states in host RAM (llama.cpp PR 16391). One saved prompt of 1521 +tokens costs 166.377 MiB. That is 112 kiB per token, exactly Qwen3-1.7B's KV +footprint (28 layers x 2 x 1024 dims x 2 bytes). + +RSS at rest, and per distinct prompt: + +| Prompts served | RSS, default | RSS, `--cache-ram 512` | +|---|---|---| +| 0 (just loaded) | 443 MB | 411 MB | +| 1 | 445 MB | 411 MB | +| 4 | 958 MB | 929 MB | +| 8 | 1641 MB | 932 MB | + +Uncapped, RSS climbs about 170MB per distinct prompt and does not stop until +the 8GiB limit. Capped at 512 MiB it plateaus at 932MB from the fourth prompt +on, with the cache holding steady at `3 prompts, 499.132 MiB` and evicting. + +The 7.9GB on the running daemon was that climb, weeks of it. Its `smaps` showed +one 6.03GB anonymous mapping at 5.32GB resident plus a 1.45GB mapping at 1.27GB +resident, and only 30MB of file-backed RSS. + +## The task's leading guess was wrong + +`-ngl 99` on the Vega iGPU costs almost no process RSS. A freshly loaded server +has 95MB of anonymous RSS in total. RADV allocates device memory through the +kernel, outside the process, and the log sees 8202 MiB free on `Vulkan0`. The +weights are mmapped and file-backed, so they are evictable and do not pin RSS. The logit buffer is not visible in the numbers above at all. + +## Decision + +`--cache-ram 512` is now the default, wired as `phraser.cache_ram_mib` and set +in `deploy/mavend.json`. 512 MiB caps total RSS near 1GB, an eighth of what the +box carried. It still holds three of the 1521-token probes above. Maven's real +routing and phrasing prompts are much shorter, so it holds more of those than +the table suggests. `-c 4096` is untouched, as #499 +required. A negative `cache_ram_mib` passes no flag, for a llama-server too old +to know it. + +Not changed: `n_parallel = 4`. With `kv_unified = true` the four slots share one +4096-token KV cache, so they do not multiply it. + +The other half of #499 was that none of these lines were reachable. mavend +scraped llama-server's stderr for the listen line and discarded it, and never +piped stdout at all. Both streams now go to mavend's log with a `llama:` prefix. +The last 12 startup lines go into the error when the server dies before it +listens. diff --git a/internal/config/config.go b/internal/config/config.go index c34d8c2..ff72b31 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -1276,6 +1276,12 @@ type PhraserConfig struct { NCtx int `json:"n_ctx,omitempty"` Timeout Duration `json:"timeout,omitempty"` + // CacheRAMMiB bounds llama-server's prompt cache. Omitted ⇒ 512 MiB, which + // is what keeps the resident model near 1 GB of RSS instead of the 7.9 GB + // measured on 2026-08-03. Set it to -1 to pass no flag at all and let the + // server apply its own 8 GiB default. See phraser.Config.CacheRAMMiB. + CacheRAMMiB int `json:"cache_ram_mib,omitempty"` + // LLMNudges — let the model word nudges again. Off by default: nudges are // worded from hand-written Russian templates now (the model broke the // persona and invented units). Chat, query and reminder phrasing always go diff --git a/internal/phraser/llmphraser.go b/internal/phraser/llmphraser.go index d0e79bf..2c5e28d 100644 --- a/internal/phraser/llmphraser.go +++ b/internal/phraser/llmphraser.go @@ -133,7 +133,7 @@ func DefaultConfig(modelPath string) Config { NCtx: 2048, // 512 MiB caps total RSS near 1 GB and still holds several recent prompts. CacheRAMMiB: 512, - Timeout: 30 * time.Second, + Timeout: 30 * time.Second, } }