fix(kernel,server,tui,inference): generic resource gauge + F-002/F-003/F-004

- generic SystemResourceProbe: owner-agnostic /proc/meminfo system RAM
  overlaid on the vendor GPU probe, wired on both managed and static paths
  so the VRAM/GPU/RAM gauge renders with an external llama-server (was dormant)
- F-002: classify provider 4xx (e.g. llama.cpp 400) as terminal, not retryable
  (except 408/429); stops retrying deterministic request failures to exhaustion
- F-003: FileSystemPromptLoader expands leading ~ / ~/ to user home
- F-004: map InferenceFailedEvent + RetryAttemptedEvent to new
  ServerMessage.InferenceFailed / RetryAttempted, decoded+rendered in tui-go;
  ArtifactContentStoredEvent explicitly mapped to null (internal, no warn)
- tests: tilde expansion, SystemResourceProbe (static/managed/fail-soft)
This commit is contained in:
2026-06-10 11:07:08 +04:00
parent a92b5a3531
commit ee24b7786a
14 changed files with 254 additions and 24 deletions
+2
View File
@@ -183,6 +183,8 @@ type Model struct {
gpuTotalMB *int64
gpuUtil *int
ramMB *int64
sysRamUsedMB *int64
sysRamTotalMB *int64
// diff / overlay
overlay OverlayKind
+13
View File
@@ -3,6 +3,7 @@ package app
import (
"crypto/rand"
"encoding/hex"
"fmt"
"strings"
"time"
@@ -172,6 +173,16 @@ func (m *Model) applyServer(msg protocol.ServerMessage) {
s.Active = false
s.addEvent(nowMillis(), "InferenceTimedOut", msg.StageID)
}
case protocol.TypeInferenceFailed:
if s := m.session(msg.SessionID); s != nil {
s.Active = false
s.addEvent(nowMillis(), "InferenceFailed", msg.StageID+": "+msg.Reason)
}
case protocol.TypeInferenceRetry:
if s := m.session(msg.SessionID); s != nil {
s.addEvent(nowMillis(), "RetryAttempted",
fmt.Sprintf("%s (%d/%d): %s", msg.StageID, msg.AttemptNumber, msg.MaxAttempts, msg.FailureReason))
}
case protocol.TypeToolStarted:
if s := m.session(msg.SessionID); s != nil {
s.Active = true
@@ -274,6 +285,8 @@ func (m *Model) applyServer(msg protocol.ServerMessage) {
m.gpuTotalMB = msg.GpuMemoryTotalMb
m.gpuUtil = msg.GpuUtilizationPct
m.ramMB = msg.ProcessRssMb
m.sysRamUsedMB = msg.SystemRamUsedMb
m.sysRamTotalMB = msg.SystemRamTotalMb
case protocol.TypeArtifactList:
m.artifacts = msg.Artifacts
m.artifactsFor = msg.SessionID
+4 -1
View File
@@ -117,8 +117,11 @@ func (m Model) gaugeText() string {
if m.gpuUtil != nil {
parts = append(parts, "GPU "+itoa(*m.gpuUtil)+"%")
}
if m.sysRamUsedMB != nil && m.sysRamTotalMB != nil {
parts = append(parts, "RAM "+itoa(int(*m.sysRamUsedMB))+"/"+itoa(int(*m.sysRamTotalMB))+"M")
}
if m.ramMB != nil {
parts = append(parts, "RAM "+itoa(int(*m.ramMB))+"M")
parts = append(parts, "proc "+itoa(int(*m.ramMB))+"M")
}
return strings.Join(parts, " ")
}
+10
View File
@@ -26,6 +26,8 @@ const (
TypeInferenceStarted = "inference.started"
TypeInferenceDone = "inference.completed"
TypeInferenceTimeout = "inference.timed_out"
TypeInferenceFailed = "inference.failed"
TypeInferenceRetry = "inference.retry"
TypeToolStarted = "tool.started"
TypeToolCompleted = "tool.completed"
TypeToolFailed = "tool.failed"
@@ -77,6 +79,11 @@ type ServerMessage struct {
Outcome string `json:"outcome"` // approval.resolved: APPROVED | REJECTED | AUTO_APPROVED
WorkspaceRoot string `json:"workspaceRoot"` // session.workspace_bound: the bound cwd
// inference.retry
AttemptNumber int `json:"attemptNumber"`
MaxAttempts int `json:"maxAttempts"`
FailureReason string `json:"failureReason"`
SteeringEmitted bool `json:"steeringEmitted"`
OccurredAt int64 `json:"occurredAt"`
ElapsedMs int64 `json:"elapsedMs"`
@@ -99,6 +106,8 @@ type ServerMessage struct {
GpuMemoryTotalMb *int64 `json:"gpuMemoryTotalMb"`
GpuUtilizationPct *int `json:"gpuUtilizationPct"`
ProcessRssMb *int64 `json:"processRssMb"`
SystemRamUsedMb *int64 `json:"systemRamUsedMb"`
SystemRamTotalMb *int64 `json:"systemRamTotalMb"`
State *SessionStateDto `json:"state"`
RiskSummary *RiskSummaryDto `json:"riskSummary"`
@@ -209,6 +218,7 @@ func (m ServerMessage) IsEventBearing() bool {
TypeSessionCompleted, TypeSessionFailed,
TypeStageStarted, TypeStageCompleted, TypeStageFailed,
TypeInferenceStarted, TypeInferenceDone, TypeInferenceTimeout,
TypeInferenceFailed, TypeInferenceRetry,
TypeToolStarted, TypeToolCompleted, TypeToolFailed, TypeToolRejected,
TypeToolAssessed,
TypeApprovalRequired, TypeApprovalResolved, TypeWorkspaceBound, TypeArtifactCreated,