diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt index ce5c8d11..e5f40479 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt @@ -74,6 +74,7 @@ import com.correx.infrastructure.inference.commons.ManagedInferenceRouter import com.correx.infrastructure.inference.commons.AmdResourceProbe import com.correx.infrastructure.inference.commons.NvidiaResourceProbe import com.correx.infrastructure.inference.commons.ResourceProbe +import com.correx.infrastructure.inference.commons.SystemResourceProbe import com.correx.infrastructure.inference.commons.UnavailableProbe import com.correx.core.inference.InferenceProvider import com.correx.infrastructure.inference.llama.cpp.LlamaCppInferenceProvider @@ -119,8 +120,10 @@ fun main() { val inferenceRouter: InferenceRouter // Non-null only on the managed path; backs manual model swap/pin from clients. var modelSwapper: ManagedInferenceRouter? = null - // Live GPU/RAM gauge; NVIDIA-backed on the managed path when nvidia-smi is present, else unavailable. - var resourceProbe: ResourceProbe = UnavailableProbe + // Live GPU/RAM gauge. The managed pid lets the GPU probe also report the model's process RSS; + // it stays null on the static path. System RAM (overlaid below) is owner-agnostic, so the gauge + // renders on both paths regardless of whether correx manages the model process. + var managedPidSupplier: () -> Long? = { null } if (correxConfig.models.isNotEmpty()) { val settings = correxConfig.modelsSettings @@ -146,13 +149,7 @@ fun main() { val managedRouter = ManagedInferenceRouter(modelManager, descriptors, targetModelConfig.id) inferenceRouter = managedRouter modelSwapper = managedRouter - resourceProbe = when { - NvidiaResourceProbe.isAvailable() -> - NvidiaResourceProbe(pidSupplier = { modelManager.currentPid() }) - AmdResourceProbe.isAvailable() -> - AmdResourceProbe(pidSupplier = { modelManager.currentPid() }) - else -> UnavailableProbe - } + managedPidSupplier = { modelManager.currentPid() } // Shutdown hook to kill the managed llama-server Runtime.getRuntime().addShutdownHook(Thread { log.info("Shutdown: unloading managed model '{}'", targetModelConfig.id) @@ -169,6 +166,15 @@ fun main() { inferenceRouter = DefaultInferenceRouter(infraRegistry, FirstAvailableRoutingStrategy()) } + // Build the resource gauge for whichever GPU vendor is present (pid-less GPU/VRAM reads work on + // both paths), then overlay owner-agnostic system RAM so the gauge always renders (F-001 gauge). + val gpuProbe: ResourceProbe = when { + NvidiaResourceProbe.isAvailable() -> NvidiaResourceProbe(pidSupplier = managedPidSupplier) + AmdResourceProbe.isAvailable() -> AmdResourceProbe(pidSupplier = managedPidSupplier) + else -> UnavailableProbe + } + val resourceProbe: ResourceProbe = SystemResourceProbe(gpuProbe) + val modelId = firstProvider.id.value logModelInfo(modelId, infraRegistry) diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt index 857b9ac8..cee9eae2 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt @@ -8,14 +8,17 @@ import com.correx.apps.server.protocol.toDto import com.correx.core.artifactstore.ArtifactStore import com.correx.core.events.events.ApprovalDecisionResolvedEvent import com.correx.core.events.events.ApprovalRequestedEvent +import com.correx.core.events.events.ArtifactContentStoredEvent import com.correx.core.events.events.ArtifactCreatedEvent import com.correx.core.events.events.ChatSessionStartedEvent import com.correx.core.events.events.ChatTurnEvent import com.correx.core.events.events.InferenceCompletedEvent import com.correx.core.events.events.SessionWorkspaceBoundEvent import com.correx.core.events.events.WorkflowStartedEvent +import com.correx.core.events.events.InferenceFailedEvent import com.correx.core.events.events.InferenceStartedEvent import com.correx.core.events.events.InferenceTimeoutEvent +import com.correx.core.events.events.RetryAttemptedEvent import com.correx.core.events.events.ModelLoadedEvent import com.correx.core.events.events.ModelUnloadedEvent import com.correx.core.events.events.OrchestrationPausedEvent @@ -149,6 +152,22 @@ suspend fun domainEventToServerMessage( sequence = seq, sessionSequence = sessionSequence, ) + is InferenceFailedEvent -> ServerMessage.InferenceFailed( + sessionId = p.sessionId, + stageId = p.stageId, + reason = p.reason, + sequence = seq, + sessionSequence = sessionSequence, + ) + is RetryAttemptedEvent -> ServerMessage.RetryAttempted( + sessionId = p.sessionId, + stageId = p.stageId, + attemptNumber = p.attemptNumber, + maxAttempts = p.maxAttempts, + failureReason = p.failureReason, + sequence = seq, + sessionSequence = sessionSequence, + ) is ToolInvocationRequestedEvent -> ServerMessage.ToolStarted( sessionId = p.sessionId, @@ -232,6 +251,9 @@ suspend fun domainEventToServerMessage( providerId = p.providerId.value, loaded = false, ) + // Internal slot→CAS-hash bookkeeping (F-007 durable bridge); no operator-facing surface. + is ArtifactContentStoredEvent -> null + else -> { log.debug( "DomainEventMapper: unmapped payload type={} sessionId={} sequence={}", diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt index dc481f35..1258cb94 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt @@ -191,6 +191,28 @@ sealed interface ServerMessage { override val sessionSequence: Long, ) : ServerMessage, SessionMessage + @Serializable + @SerialName("inference.failed") + data class InferenceFailed( + val sessionId: SessionId, + val stageId: StageId, + val reason: String, + override val sequence: Long, + override val sessionSequence: Long, + ) : ServerMessage, SessionMessage + + @Serializable + @SerialName("inference.retry") + data class RetryAttempted( + val sessionId: SessionId, + val stageId: StageId, + val attemptNumber: Int, + val maxAttempts: Int, + val failureReason: String, + override val sequence: Long, + override val sessionSequence: Long, + ) : ServerMessage, SessionMessage + // -- Tool execution -- @Serializable @@ -336,6 +358,8 @@ sealed interface ServerMessage { val gpuMemoryTotalMb: Long?, val gpuUtilizationPct: Int?, val processRssMb: Long?, + val systemRamUsedMb: Long? = null, + val systemRamTotalMb: Long? = null, override val sequence: Long? = null, override val sessionSequence: Long? = null, ) : ServerMessage, NonEventMessage diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/ws/GlobalStreamHandler.kt b/apps/server/src/main/kotlin/com/correx/apps/server/ws/GlobalStreamHandler.kt index ae0ce09f..df499216 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/ws/GlobalStreamHandler.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/ws/GlobalStreamHandler.kt @@ -61,6 +61,8 @@ private fun ResourceSnapshot.toResourceStatus(): ServerMessage.ResourceStatus = gpuMemoryTotalMb = gpu?.memoryTotalMb, gpuUtilizationPct = gpu?.utilizationPct, processRssMb = processRssBytes?.let { it / BYTES_PER_MB }, + systemRamUsedMb = systemRamUsedBytes?.let { it / BYTES_PER_MB }, + systemRamTotalMb = systemRamTotalBytes?.let { it / BYTES_PER_MB }, ) class GlobalStreamHandler(private val module: ServerModule) { diff --git a/apps/tui-go/internal/app/model.go b/apps/tui-go/internal/app/model.go index 8eff9aaa..883913b0 100644 --- a/apps/tui-go/internal/app/model.go +++ b/apps/tui-go/internal/app/model.go @@ -183,6 +183,8 @@ type Model struct { gpuTotalMB *int64 gpuUtil *int ramMB *int64 + sysRamUsedMB *int64 + sysRamTotalMB *int64 // diff / overlay overlay OverlayKind diff --git a/apps/tui-go/internal/app/server.go b/apps/tui-go/internal/app/server.go index 0bb982ba..5cf66f7a 100644 --- a/apps/tui-go/internal/app/server.go +++ b/apps/tui-go/internal/app/server.go @@ -3,6 +3,7 @@ package app import ( "crypto/rand" "encoding/hex" + "fmt" "strings" "time" @@ -172,6 +173,16 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { s.Active = false s.addEvent(nowMillis(), "InferenceTimedOut", msg.StageID) } + case protocol.TypeInferenceFailed: + if s := m.session(msg.SessionID); s != nil { + s.Active = false + s.addEvent(nowMillis(), "InferenceFailed", msg.StageID+": "+msg.Reason) + } + case protocol.TypeInferenceRetry: + if s := m.session(msg.SessionID); s != nil { + s.addEvent(nowMillis(), "RetryAttempted", + fmt.Sprintf("%s (%d/%d): %s", msg.StageID, msg.AttemptNumber, msg.MaxAttempts, msg.FailureReason)) + } case protocol.TypeToolStarted: if s := m.session(msg.SessionID); s != nil { s.Active = true @@ -274,6 +285,8 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { m.gpuTotalMB = msg.GpuMemoryTotalMb m.gpuUtil = msg.GpuUtilizationPct m.ramMB = msg.ProcessRssMb + m.sysRamUsedMB = msg.SystemRamUsedMb + m.sysRamTotalMB = msg.SystemRamTotalMb case protocol.TypeArtifactList: m.artifacts = msg.Artifacts m.artifactsFor = msg.SessionID diff --git a/apps/tui-go/internal/app/view.go b/apps/tui-go/internal/app/view.go index 0dd4888f..721a90fa 100644 --- a/apps/tui-go/internal/app/view.go +++ b/apps/tui-go/internal/app/view.go @@ -117,8 +117,11 @@ func (m Model) gaugeText() string { if m.gpuUtil != nil { parts = append(parts, "GPU "+itoa(*m.gpuUtil)+"%") } + if m.sysRamUsedMB != nil && m.sysRamTotalMB != nil { + parts = append(parts, "RAM "+itoa(int(*m.sysRamUsedMB))+"/"+itoa(int(*m.sysRamTotalMB))+"M") + } if m.ramMB != nil { - parts = append(parts, "RAM "+itoa(int(*m.ramMB))+"M") + parts = append(parts, "proc "+itoa(int(*m.ramMB))+"M") } return strings.Join(parts, " ") } diff --git a/apps/tui-go/internal/protocol/protocol.go b/apps/tui-go/internal/protocol/protocol.go index 8db59bed..87bd31ac 100644 --- a/apps/tui-go/internal/protocol/protocol.go +++ b/apps/tui-go/internal/protocol/protocol.go @@ -26,6 +26,8 @@ const ( TypeInferenceStarted = "inference.started" TypeInferenceDone = "inference.completed" TypeInferenceTimeout = "inference.timed_out" + TypeInferenceFailed = "inference.failed" + TypeInferenceRetry = "inference.retry" TypeToolStarted = "tool.started" TypeToolCompleted = "tool.completed" TypeToolFailed = "tool.failed" @@ -77,6 +79,11 @@ type ServerMessage struct { Outcome string `json:"outcome"` // approval.resolved: APPROVED | REJECTED | AUTO_APPROVED WorkspaceRoot string `json:"workspaceRoot"` // session.workspace_bound: the bound cwd + // inference.retry + AttemptNumber int `json:"attemptNumber"` + MaxAttempts int `json:"maxAttempts"` + FailureReason string `json:"failureReason"` + SteeringEmitted bool `json:"steeringEmitted"` OccurredAt int64 `json:"occurredAt"` ElapsedMs int64 `json:"elapsedMs"` @@ -99,6 +106,8 @@ type ServerMessage struct { GpuMemoryTotalMb *int64 `json:"gpuMemoryTotalMb"` GpuUtilizationPct *int `json:"gpuUtilizationPct"` ProcessRssMb *int64 `json:"processRssMb"` + SystemRamUsedMb *int64 `json:"systemRamUsedMb"` + SystemRamTotalMb *int64 `json:"systemRamTotalMb"` State *SessionStateDto `json:"state"` RiskSummary *RiskSummaryDto `json:"riskSummary"` @@ -209,6 +218,7 @@ func (m ServerMessage) IsEventBearing() bool { TypeSessionCompleted, TypeSessionFailed, TypeStageStarted, TypeStageCompleted, TypeStageFailed, TypeInferenceStarted, TypeInferenceDone, TypeInferenceTimeout, + TypeInferenceFailed, TypeInferenceRetry, TypeToolStarted, TypeToolCompleted, TypeToolFailed, TypeToolRejected, TypeToolAssessed, TypeApprovalRequired, TypeApprovalResolved, TypeWorkspaceBound, TypeArtifactCreated, diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt index a25b4136..fa64af7f 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt @@ -133,6 +133,10 @@ private const val STAGE_COMPLETE_TOOL = "stage_complete" private const val OUTPUT_SUMMARY_LIMIT = 500 private const val REPO_MAP_INJECT_TOP_K = 30 +// HTTP statuses that are transient despite being 4xx (F-002 retry classification). +private const val HTTP_TIMEOUT = 408 +private const val HTTP_TOO_MANY_REQUESTS = 429 + @SuppressWarnings( "ForbiddenComment", "UnusedParameter", @@ -474,7 +478,8 @@ abstract class SessionOrchestrator( } return when (inferenceResult) { is InferenceResult.Cancelled -> StageExecutionResult.Failure("CANCELLED", retryable = false) - is InferenceResult.Failed -> StageExecutionResult.Failure(inferenceResult.reason, retryable = true) + is InferenceResult.Failed -> + StageExecutionResult.Failure(inferenceResult.reason, retryable = isRetryableFailure(inferenceResult.reason)) is InferenceResult.Success -> { if (isCancelled(sessionId)) return StageExecutionResult.Failure("CANCELLED", retryable = false) // Capture the LLM-emitted artifact's JSON so transition conditions @@ -1329,6 +1334,16 @@ abstract class SessionOrchestrator( } } + // A provider 4xx client error (e.g. llama.cpp 400 "invalid_request") is a deterministic + // request-construction failure — retrying re-sends the identical bad request and exhausts the + // budget for nothing (F-002). Treat 4xx as terminal, except 408 (timeout) and 429 (rate-limit) + // which are transient. Everything else (network/5xx/unknown) stays retryable. + private fun isRetryableFailure(reason: String): Boolean { + val status = Regex("""returned\s+(\d{3})""").find(reason)?.groupValues?.get(1)?.toIntOrNull() + ?: return true + return status !in 400..499 || status == HTTP_TIMEOUT || status == HTTP_TOO_MANY_REQUESTS + } + // --- transition helpers --- internal fun resolveTransition( diff --git a/infrastructure/inference/commons/src/main/kotlin/com/correx/infrastructure/inference/commons/ResourceProbe.kt b/infrastructure/inference/commons/src/main/kotlin/com/correx/infrastructure/inference/commons/ResourceProbe.kt index 6efad31e..754d2529 100644 --- a/infrastructure/inference/commons/src/main/kotlin/com/correx/infrastructure/inference/commons/ResourceProbe.kt +++ b/infrastructure/inference/commons/src/main/kotlin/com/correx/infrastructure/inference/commons/ResourceProbe.kt @@ -1,5 +1,8 @@ package com.correx.infrastructure.inference.commons +import org.slf4j.LoggerFactory +import java.io.File + /** * Whole-device GPU usage at an instant. All fields are device-wide (not per-process) — * per-model VRAM attribution is an explicit non-goal. @@ -11,13 +14,19 @@ data class GpuStatus( ) /** - * A point-in-time reading of local-model resource usage. Both fields are nullable so that - * "unavailable" is representable without exceptions: [gpu] is null when no GPU probe is present - * (non-NVIDIA box / `nvidia-smi` absent), [processRssBytes] is null when no model is resident. + * A point-in-time reading of local resource usage. All fields are nullable so that "unavailable" + * is representable without exceptions: [gpu] is null when no GPU probe is present (non-NVIDIA/AMD + * box or the tool is absent), [processRssBytes] is null when no correx-managed model is resident, + * and [systemRamUsedBytes]/[systemRamTotalBytes] are null when `/proc/meminfo` can't be read. + * + * System RAM is **owner-agnostic** — it reflects the whole host regardless of who launched the + * model, so the gauge renders even on the static (external `llama-server`) path. */ data class ResourceSnapshot( val gpu: GpuStatus?, val processRssBytes: Long?, + val systemRamUsedBytes: Long? = null, + val systemRamTotalBytes: Long? = null, ) { companion object { val UNAVAILABLE = ResourceSnapshot(gpu = null, processRssBytes = null) @@ -34,7 +43,49 @@ fun interface ResourceProbe { fun probe(): ResourceSnapshot } -/** Always-unavailable probe — used on non-GPU boxes and the static (unmanaged) provider path. */ +/** Always-unavailable probe — used on non-GPU boxes when even system RAM can't be read. */ object UnavailableProbe : ResourceProbe { override fun probe(): ResourceSnapshot = ResourceSnapshot.UNAVAILABLE } + +/** + * Owner-agnostic composite probe: delegates GPU/process-RSS to a vendor [gpuProbe] (NVIDIA/AMD or + * [UnavailableProbe]) and always overlays whole-host system RAM read from `/proc/meminfo`. This is + * what makes the gauge render on BOTH the managed and static (external `llama-server`) paths — the + * system-RAM read needs no managed pid. The RAM read is injectable for testing. + */ +class SystemResourceProbe( + private val gpuProbe: ResourceProbe = UnavailableProbe, + private val systemRamReader: () -> Pair? = SystemRam::readUsedTotalBytes, +) : ResourceProbe { + override fun probe(): ResourceSnapshot { + val base = gpuProbe.probe() + val ram = systemRamReader() + return base.copy( + systemRamUsedBytes = ram?.first, + systemRamTotalBytes = ram?.second, + ) + } +} + +/** Reads whole-host RAM from `/proc/meminfo`. Fails soft to null so a non-Linux host never throws. */ +object SystemRam { + private val log = LoggerFactory.getLogger("SystemRam") + private const val BYTES_PER_KB = 1024L + + /** Returns (usedBytes, totalBytes) where used = MemTotal − MemAvailable, or null if unreadable. */ + fun readUsedTotalBytes(): Pair? = runCatching { + val meminfo = File("/proc/meminfo") + if (!meminfo.exists()) return null + val lines = meminfo.readLines() + val total = lines.kbValue("MemTotal:") ?: return null + val available = lines.kbValue("MemAvailable:") ?: return null + val used = (total - available).coerceAtLeast(0) + (used * BYTES_PER_KB) to (total * BYTES_PER_KB) + }.onFailure { log.debug("/proc/meminfo read failed: {}", it.message) }.getOrNull() + + // Parses a "Key: 12345 kB" line to its kB value. + private fun List.kbValue(key: String): Long? = + firstOrNull { it.startsWith(key) } + ?.removePrefix(key)?.trim()?.substringBefore(' ')?.toLongOrNull() +} diff --git a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoader.kt b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoader.kt index 2df760ed..8a3522d3 100644 --- a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoader.kt +++ b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoader.kt @@ -8,12 +8,21 @@ class FileSystemPromptLoader( private val configDir: Path = Path.of(System.getProperty("user.home"), ".config", "correx", "prompts"), ) : PromptLoader { override fun load(path: String): String { - val local = Path.of(path) + val expanded = expandTilde(path) + val local = Path.of(expanded) if (local.exists()) return local.readText() - val fromConfig = configDir.resolve(path) + val fromConfig = configDir.resolve(expanded) if (fromConfig.exists()) return fromConfig.readText() throw PromptNotFoundException("Prompt not found: $path (searched: $local, $fromConfig)") } + + // Expand a leading `~` / `~/...` to the user home so a config-supplied home path resolves + // instead of being concatenated literally onto [configDir] (F-003). + private fun expandTilde(path: String): String = when { + path == "~" -> System.getProperty("user.home") + path.startsWith("~/") -> System.getProperty("user.home") + path.substring(1) + else -> path + } } diff --git a/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoaderTest.kt b/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoaderTest.kt index 6ff750ad..1585fef1 100644 --- a/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoaderTest.kt +++ b/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/FileSystemPromptLoaderTest.kt @@ -33,6 +33,19 @@ class FileSystemPromptLoaderTest { assertContains(ex.message!!, configDir.toString()) } + @Test + fun `expands leading tilde to user home (F-003)`() { + val home = System.getProperty("user.home") + val rel = "correx-tilde-${System.nanoTime()}.md" + val file = java.nio.file.Path.of(home, rel).also { it.writeText("home prompt") } + try { + val loader = FileSystemPromptLoader(configDir = Files.createTempDirectory("unused")) + assertEquals("home prompt", loader.load("~/$rel")) + } finally { + Files.deleteIfExists(file) + } + } + @Test fun `local path takes precedence over configDir`() { val localFile = Files.createTempFile("prompt", ".md").also { it.writeText("local") } diff --git a/qa/audit-report-2026-06-08.md b/qa/audit-report-2026-06-08.md index ee751cbe..f361727c 100644 --- a/qa/audit-report-2026-06-08.md +++ b/qa/audit-report-2026-06-08.md @@ -94,7 +94,9 @@ repo `docs/schemas/` were stale relative to live — a clean-checkout reproducti identical 400s within ~80ms, then `WorkflowFailedEvent retryExhausted=true`. - **suspected cause:** retry classifier treats all provider failures as retryable. A 400 `invalid_request_error` is deterministic request-construction failure → should be terminal. -- **status:** confirmed +- **status:** FIXED (2026-06-10) — `SessionOrchestrator.isRetryableFailure` parses the + provider status from the failure reason; 4xx is terminal (except 408/429), 5xx/network/ + unknown stay retryable. So a 400 now fails fast instead of burning the retry budget. ### F-003 — `~` not expanded in default_system prompt path - **subsystem:** core/kernel (prompt loading) / core/config @@ -103,7 +105,8 @@ repo `docs/schemas/` were stale relative to live — a clean-checkout reproducti Prompt not found (searched: .../prompts/~/.config/correx/prompts/default_system.md)` — tilde concatenated literally onto the config dir. Logged ERROR on every attempt. Non-fatal (run proceeds) but noisy + the default system prompt is silently absent. -- **status:** confirmed +- **status:** FIXED (2026-06-10) — `FileSystemPromptLoader.expandTilde` resolves a leading + `~`/`~/` to the user home before lookup (unit-tested). ### F-005 — file_read jails relative paths against process cwd, not the workspace - **invariant/contract:** functional (tool jailing) / two-plane consistency (#9) @@ -192,9 +195,10 @@ repo `docs/schemas/` were stale relative to live — a clean-checkout reproducti - **evidence:** `DomainEventMapper: unmapped payload type=InferenceFailedEvent` and `=RetryAttemptedEvent` repeated. Operator sees failure only via router narration, not a structured failure/retry surface in the TUI. -- **status:** confirmed -- **note:** `ArtifactContentStoredEvent` is also unmapped (`DomainEventMapper: unmapped - payload type=ArtifactContentStoredEvent`) — same class of gap, new event added this session. +- **status:** FIXED (2026-06-10) — new `ServerMessage.InferenceFailed` / + `RetryAttempted` variants; `DomainEventMapper` arms map `InferenceFailedEvent` / + `RetryAttemptedEvent`; tui-go decodes + renders them as session events. The internal + `ArtifactContentStoredEvent` is now explicitly mapped to null (no operator surface, no warn). ### F-008 — stage `token_budget` not propagated to provider `max_tokens` → truncation - **invariant/contract:** functional (workflows run end-to-end) / #7 (validatable output) @@ -326,8 +330,9 @@ repo `docs/schemas/` were stale relative to live — a clean-checkout reproducti contents) were cut off mid-generation, burning all retries → `did not produce patch`. - **root cause:** `OrchestrationConfig.stageTimeoutMs` defaults to `60_000L` and is hardcoded in `Main` (`defaultOrchestrationConfig`) with no config wiring. -- **status:** PARTIAL — raised to `180_000L` in `Main`. Should be surfaced as a config knob - (TODO left at the patch site). +- **status:** FIXED — now a real config knob `orchestration.stage_timeout_ms` + (`CorrexConfig.OrchestrationKnobs`, default 180s), parsed by `ConfigLoader`, written by + `CorrexConfigWriter`, live-editable via `EditableConfig`, wired in `Main.kt`. ### F-017 (observation, not a correx defect) — local model does not reliably issue write tools - **invariant/contract:** n/a (model capability) diff --git a/testing/contracts/src/test/kotlin/com/correx/testing/contracts/model/inference/SystemResourceProbeTest.kt b/testing/contracts/src/test/kotlin/com/correx/testing/contracts/model/inference/SystemResourceProbeTest.kt new file mode 100644 index 00000000..eea9dc98 --- /dev/null +++ b/testing/contracts/src/test/kotlin/com/correx/testing/contracts/model/inference/SystemResourceProbeTest.kt @@ -0,0 +1,55 @@ +package com.correx.testing.contracts.model.inference + +import com.correx.infrastructure.inference.commons.GpuStatus +import com.correx.infrastructure.inference.commons.ResourceProbe +import com.correx.infrastructure.inference.commons.ResourceSnapshot +import com.correx.infrastructure.inference.commons.SystemResourceProbe +import com.correx.infrastructure.inference.commons.UnavailableProbe +import org.junit.jupiter.api.Assertions.assertEquals +import org.junit.jupiter.api.Assertions.assertNull +import org.junit.jupiter.api.Test + +class SystemResourceProbeTest { + + private val bytesPerKb = 1024L + + @Test + fun `overlays system RAM onto an unavailable GPU probe (static path, F-001 gauge)`() { + val probe = SystemResourceProbe( + gpuProbe = UnavailableProbe, + systemRamReader = { (4L * bytesPerKb) to (16L * bytesPerKb) }, + ) + val snap = probe.probe() + assertNull(snap.gpu) // no managed model / no GPU, yet RAM still renders + assertNull(snap.processRssBytes) + assertEquals(4L * bytesPerKb, snap.systemRamUsedBytes) + assertEquals(16L * bytesPerKb, snap.systemRamTotalBytes) + } + + @Test + fun `preserves the delegate GPU and process RSS while adding system RAM (managed path)`() { + val gpuDelegate = ResourceProbe { + ResourceSnapshot( + gpu = GpuStatus(memoryUsedMb = 2048, memoryTotalMb = 8192, utilizationPct = 50), + processRssBytes = 999L, + ) + } + val probe = SystemResourceProbe( + gpuProbe = gpuDelegate, + systemRamReader = { 100L to 200L }, + ) + val snap = probe.probe() + assertEquals(2048L, snap.gpu?.memoryUsedMb) + assertEquals(999L, snap.processRssBytes) + assertEquals(100L, snap.systemRamUsedBytes) + assertEquals(200L, snap.systemRamTotalBytes) + } + + @Test + fun `fails soft to null system RAM when the reader is unavailable`() { + val probe = SystemResourceProbe(gpuProbe = UnavailableProbe, systemRamReader = { null }) + val snap = probe.probe() + assertNull(snap.systemRamUsedBytes) + assertNull(snap.systemRamTotalBytes) + } +}