diff --git a/.correxignore b/.correxignore new file mode 100644 index 00000000..f1a6c0d9 --- /dev/null +++ b/.correxignore @@ -0,0 +1,14 @@ +# Repo-map exclusions — silences generic-similarity noise in the semantic "Relevant files" +# feed without touching CorrexConfig's ignore_globs. One pattern per line, matched by +# RepoMapIndexer.isIgnored: bare name = matches at any depth, path with '/' = subtree prefix. + +# Dev/QA tooling, not project business logic. +scripts + +# Fixture repo used by QA gate tests — deliberately fake code, not part of correx itself. +qa/sample-repo + +# Frontend build config — boilerplate, never what a code-context query is actually after. +frontend/tailwind.config.js +frontend/vite.config.ts +frontend/postcss.config.js diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt index 4030a95b..ab0099d5 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt @@ -367,6 +367,7 @@ fun main() { embedder = embedder, l3MemoryStore = l3MemoryStore, repoRoot = workspaceRoot.toString(), + eventStore = eventStore, ) } else { null diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/memory/L3RepoKnowledgeRetriever.kt b/apps/server/src/main/kotlin/com/correx/apps/server/memory/L3RepoKnowledgeRetriever.kt index 5a817e1d..42759b4f 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/memory/L3RepoKnowledgeRetriever.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/memory/L3RepoKnowledgeRetriever.kt @@ -1,25 +1,74 @@ package com.correx.apps.server.memory +import com.correx.core.events.events.EventMetadata +import com.correx.core.events.events.NewEvent import com.correx.core.events.events.RepoKnowledgeHit +import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent +import com.correx.core.events.stores.EventStore +import com.correx.core.events.types.EventId import com.correx.core.events.types.SessionId import com.correx.core.inference.Embedder import com.correx.core.kernel.orchestration.RepoKnowledgeRetriever import com.correx.core.talkie.l3.L3MemoryStore import com.correx.core.talkie.l3.L3Query +import kotlinx.coroutines.CancellationException +import kotlinx.datetime.Clock +import org.slf4j.LoggerFactory +import java.util.UUID + +private val log = LoggerFactory.getLogger(L3RepoKnowledgeRetriever::class.java) private const val RETRIEVAL_OVERSAMPLE_FACTOR = 4 +// Real embedder scores for genuinely relevant repo-map hits cluster 0.68-0.80 (2026-07-03 +// live grounding); below this floor a hit is just the nearest thing in a small corpus, not +// actually related to the query — rendering it as "relevant" is the same poisoning failure +// mode as the markdown-in-L3 bug, just via low-signal cosine similarity instead of topic drift. +private const val MIN_SIMILARITY_SCORE = 0.5f + class L3RepoKnowledgeRetriever( private val embedder: Embedder, private val l3MemoryStore: L3MemoryStore, private val repoRoot: String, + private val eventStore: EventStore? = null, ) : RepoKnowledgeRetriever { override suspend fun retrieve(sessionId: SessionId, query: String, k: Int): List { val vector = embedder.embed(query) - return l3MemoryStore.query(L3Query(vector = vector, k = k * RETRIEVAL_OVERSAMPLE_FACTOR)) + val candidates = l3MemoryStore.query(L3Query(vector = vector, k = k * RETRIEVAL_OVERSAMPLE_FACTOR)) // Trailing ':' so "/repo" does not also match "/repo2" turnIds (prefix collision). .filter { it.entry.turnId.startsWith("repomap:$repoRoot:") } - .take(k) - .map { RepoKnowledgeHit(path = it.entry.text.substringBefore(":"), text = it.entry.text, score = it.score) } + val (kept, dropped) = candidates.partition { it.score >= MIN_SIMILARITY_SCORE } + if (dropped.isNotEmpty()) recordDropped(sessionId, query, dropped.map { it.toHit() }) + return kept.take(k).map { it.toHit() } } + + private suspend fun recordDropped(sessionId: SessionId, query: String, dropped: List) { + val store = eventStore ?: return + runCatching { + store.append( + NewEvent( + metadata = EventMetadata( + eventId = EventId(UUID.randomUUID().toString()), + sessionId = sessionId, + timestamp = Clock.System.now(), + schemaVersion = 1, + causationId = null, + correlationId = null, + ), + payload = RepoKnowledgeHitsFilteredEvent( + sessionId = sessionId, + query = query, + threshold = MIN_SIMILARITY_SCORE, + dropped = dropped, + ), + ), + ) + }.exceptionOrNull()?.let { e -> + if (e is CancellationException) throw e + log.warn("failed to record filtered repo-knowledge hits for {}: {}", sessionId.value, e.message) + } + } + + private fun com.correx.core.talkie.l3.L3Hit.toHit() = + RepoKnowledgeHit(path = entry.text.substringBefore(":"), text = entry.text, score = score) } diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/memory/ProjectMemoryService.kt b/apps/server/src/main/kotlin/com/correx/apps/server/memory/ProjectMemoryService.kt index be7b6515..924ab20f 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/memory/ProjectMemoryService.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/memory/ProjectMemoryService.kt @@ -117,7 +117,14 @@ class ProjectMemoryService( // Trailing ':' delimiter so a repoRoot prefix can't collide with a longer sibling // (/repo vs /repo2) under the retriever's startsWith filter. val tag = if (stateKey != null) "repomap:$repoRoot:$stateKey" else "repomap:$repoRoot:" - entries.forEach { entry -> + // Docs are already surfaced via the "Docs available" catalog built straight from + // RepoMapComputedEvent (SessionOrchestrator, 2026-07-07 doc-injection rework) — embedding + // them into the same L3 namespace as code let generic textual similarity (e.g. "analyst", + // "compression pipeline") outrank the actual code files for a code-focused query, silently + // crowding the semantic "Relevant files" feed with irrelevant markdown (2026-07-08 recurrence + // of the 2026-07-05 context-poisoning root cause, this time via L3RepoKnowledgeRetriever + // rather than the repo-map floor that was fixed then). + entries.filterNot { it.path.endsWith(".md", ignoreCase = true) }.forEach { entry -> runCatching { val text = entry.path + if (entry.symbols.isEmpty()) "" else ": ${entry.symbols.joinToString(", ")}" l3MemoryStore.store( diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/memory/RepoMapIndexer.kt b/apps/server/src/main/kotlin/com/correx/apps/server/memory/RepoMapIndexer.kt index 3734bcc1..3cd3e50d 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/memory/RepoMapIndexer.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/memory/RepoMapIndexer.kt @@ -32,12 +32,12 @@ class RepoMapIndexer( ) : RepoMapIndexerPort { override fun index(repoRoot: Path): List { if (!repoRoot.isDirectory()) return emptyList() - val ignore = ignoreGlobs.toSet() + val ignore = ignoreGlobs.toSet() + loadCorrexIgnore(repoRoot) val files = Files.walk(repoRoot, maxDepth).use { stream -> stream.filter { it.isRegularFile() } .filter { it.extension in SYMBOL_PATTERNS.keys } - .filter { path -> path.relativeTo(repoRoot).none { it.name in ignore } } + .filter { path -> !isIgnored(path.relativeTo(repoRoot), ignore) } .toList() } if (files.isEmpty()) return emptyList() @@ -58,6 +58,35 @@ class RepoMapIndexer( .sortedByDescending { it.score } } + /** + * A repo-local, gitignore-style opt-out for the repo map: patterns here apply on top of + * `[project] ignore_globs` (CorrexConfig) without needing a config edit, so anyone working in + * the repo can silence a noisy path for themselves. Same matching rules as [isIgnored] — one + * pattern per line, blank lines and `#` comments skipped. Missing file is not an error (most + * repos won't have one). + */ + private fun loadCorrexIgnore(repoRoot: Path): List { + val file = repoRoot.resolve(".correxignore") + if (!file.isRegularFile()) return emptyList() + return runCatching { file.readText() }.getOrNull() + ?.lines() + ?.map { it.trim() } + ?.filter { it.isNotEmpty() && !it.startsWith("#") } + ?: emptyList() + } + + /** + * An ignore entry matches either a bare path-segment name (any depth, e.g. "node_modules") or a + * relative path prefix (e.g. "examples/workflows/prompts", to exclude just that subtree). + */ + private fun isIgnored(relativePath: Path, ignore: Set): Boolean { + val rel = relativePath.toString().replace('\\', '/') + return ignore.any { pattern -> + if ('/' in pattern) rel == pattern || rel.startsWith("$pattern/") + else relativePath.any { it.name == pattern } + } + } + private fun extractSymbols(path: Path): List { // Docs are indexed as a single "what is this" descriptor, not a table-of-contents dump: a // heading flood (40 h1–h3 per doc across a large docs/ tree) buries real source and force-feeds diff --git a/apps/tui-go/internal/app/model.go b/apps/tui-go/internal/app/model.go index 96a92974..f54b679e 100644 --- a/apps/tui-go/internal/app/model.go +++ b/apps/tui-go/internal/app/model.go @@ -89,6 +89,22 @@ type EventEntry struct { Detail string } +// PlanStageStatus tracks a stage's lifecycle in a locked execution plan. +type PlanStageStatus int + +const ( + PlanPending PlanStageStatus = iota + PlanRunning + PlanCompleted + PlanFailed +) + +// PlanStage is one stage in a locked execution plan, with live execution status. +type PlanStage struct { + ID string + Status PlanStageStatus +} + // ToolStatus mirrors the Kotlin ToolDisplayStatus. type ToolStatus int @@ -194,6 +210,8 @@ type Session struct { Name string LastEventAt int64 CurrentStage string + PlanGoal string + PlanStages []PlanStage WorkspaceRoot string // bound cwd, from session.workspace_bound LastOutput string LastResponse string diff --git a/apps/tui-go/internal/app/render_matrix_test.go b/apps/tui-go/internal/app/render_matrix_test.go index f5e2edab..486ed754 100644 --- a/apps/tui-go/internal/app/render_matrix_test.go +++ b/apps/tui-go/internal/app/render_matrix_test.go @@ -612,9 +612,11 @@ var nonRenderingEventTypes = map[string]string{ protocol.TypeSnapshotComplete: "control frame: ends the snapshot-buffering phase; no render", protocol.TypeRouterResponse: "recognized no-op: superseded by chat.turn on the global stream", protocol.TypeProtocolError: "transport diagnostic: explicit no-op, not surfaced", - protocol.TypeFileList: "query reply: populates the @ file picker overlay, not the transcript", - protocol.TypeGrantList: "query reply: populates the standing-grants overlay, not the transcript", - protocol.TypeHealthChecks: "query reply: populates the health-checks overlay, not the transcript", + protocol.TypeFileList: "query reply: populates the @ file picker overlay, not the transcript", + protocol.TypeGrantList: "query reply: populates the standing-grants overlay, not the transcript", + protocol.TypeHealthChecks: "query reply: populates the health-checks overlay, not the transcript", + protocol.TypeProjectProfile: "query reply: populates the project profile overlay, not the transcript", + protocol.TypeOperatorProfile: "query reply: populates the operator profile overlay, not the transcript", } // TestRenderMatrixCoversEveryEventType is the load-bearing matrix guarantee: it scans diff --git a/apps/tui-go/internal/app/server.go b/apps/tui-go/internal/app/server.go index ce0f4e67..0e97fe3d 100644 --- a/apps/tui-go/internal/app/server.go +++ b/apps/tui-go/internal/app/server.go @@ -153,16 +153,34 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { s.Tools = nil s.StageTokensUsed = 0 s.addEvent(msg.OccurredAt, "StageStarted", msg.StageID) + for i := range s.PlanStages { + if s.PlanStages[i].ID == msg.StageID { + s.PlanStages[i].Status = PlanRunning + break + } + } } case protocol.TypeStageCompleted: if s := m.session(msg.SessionID); s != nil { s.CurrentStage = "" s.addEvent(msg.OccurredAt, "StageCompleted", msg.StageID) + for i := range s.PlanStages { + if s.PlanStages[i].ID == msg.StageID { + s.PlanStages[i].Status = PlanCompleted + break + } + } } case protocol.TypeStageFailed: if s := m.session(msg.SessionID); s != nil { s.CurrentStage = "" s.addEvent(msg.OccurredAt, "StageFailed", msg.StageID) + for i := range s.PlanStages { + if s.PlanStages[i].ID == msg.StageID { + s.PlanStages[i].Status = PlanFailed + break + } + } } case protocol.TypeInferenceStarted: if s := m.session(msg.SessionID); s != nil { @@ -303,6 +321,11 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { if s := m.session(msg.SessionID); s != nil { s.addEvent(nowMillis(), "PlanLocked", strings.Join(msg.StageIDs, " → ")) s.LastOutput = "plan locked: " + strings.Join(msg.StageIDs, " → ") + s.PlanGoal = msg.Content + s.PlanStages = make([]PlanStage, 0, len(msg.StageIDs)) + for _, id := range msg.StageIDs { + s.PlanStages = append(s.PlanStages, PlanStage{ID: id, Status: PlanPending}) + } } case protocol.TypeWorkspaceBound: if s := m.session(msg.SessionID); s != nil { diff --git a/apps/tui-go/internal/app/view.go b/apps/tui-go/internal/app/view.go index e31f0850..3595089b 100644 --- a/apps/tui-go/internal/app/view.go +++ b/apps/tui-go/internal/app/view.go @@ -790,6 +790,10 @@ func (m Model) buildTranscriptRows(w int) ([]string, []int) { t := m.theme msgs := m.routerMessages[m.selectedID] var rows []string + // Prepend the live execution plan card when the session has locked plan stages. + if planRows := m.planCardLines(); len(planRows) > 0 { + rows = append(rows, planRows...) + } msgStart := make([]int, len(msgs)) // first row index of each message, for scroll-to-selection for mi, e := range msgs { msgStart[mi] = len(rows) @@ -847,6 +851,53 @@ func (m Model) buildTranscriptRows(w int) ([]string, []int) { return rows, msgStart } +// planCardLines renders the live execution plan card for the selected session. +func (m Model) planCardLines() []string { + t := m.theme + s := m.session(m.selectedID) + if s == nil || len(s.PlanStages) == 0 { + return nil + } + goal := strings.TrimSpace(s.PlanGoal) + var out []string + if goal != "" { + out = append(out, t.span(goal, t.P.Faint)) + } + stageLabel := "stage" + if len(s.PlanStages) != 1 { + stageLabel = "stages" + } + out = append(out, t.span("Plan \u00b7 "+itoa(len(s.PlanStages))+" "+stageLabel, t.P.Accent)) + idW := 0 + for _, ps := range s.PlanStages { + if len(ps.ID) > idW { + idW = len(ps.ID) + } + } + for _, ps := range s.PlanStages { + var badge, badgeStyle string + var badgeColor color.Color + switch ps.Status { + case PlanPending: + badge = "\u25cb" + badgeColor = t.P.Faint + case PlanRunning: + badge = "\u25cf" + badgeColor = t.P.Accent + case PlanCompleted: + badge = "\u2713" + badgeColor = t.P.OK + case PlanFailed: + badge = "\u2717" + badgeColor = t.P.Bad + } + badgeStyle = t.span(badge, badgeColor) + rest := t.span(" "+padRaw(ps.ID, idW), t.P.Fg) + out = append(out, " "+badgeStyle+rest) + } + return out +} + func (m Model) eventRows(w, h int) []string { t := m.theme header := lipgloss.NewStyle().Foreground(t.P.Accent).Background(t.P.Bg).Bold(true).Render("event stream") + diff --git a/core/config/src/main/kotlin/com/correx/core/config/CorrexConfig.kt b/core/config/src/main/kotlin/com/correx/core/config/CorrexConfig.kt index 04fe4060..1b529324 100644 --- a/core/config/src/main/kotlin/com/correx/core/config/CorrexConfig.kt +++ b/core/config/src/main/kotlin/com/correx/core/config/CorrexConfig.kt @@ -94,6 +94,10 @@ data class ProjectConfig( companion object { val DEFAULT_IGNORES: List = listOf( ".git", "node_modules", "build", "target", ".gradle", "dist", ".idea", ".opencode", + // Role-prompt templates: loaded by workflow config as LLM system prompts, not project + // docs — indexing them lets a stage `file_read` another role's prompt as if it were + // documentation. + "examples/workflows/prompts", ) } } diff --git a/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt b/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt index 87dd2b22..03499627 100644 --- a/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt +++ b/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt @@ -26,6 +26,9 @@ import com.correx.core.events.types.ContextEntryId import com.correx.core.events.types.ContextPackId import com.correx.core.events.types.SessionId import com.correx.core.events.types.StageId +import kotlinx.serialization.json.Json +import kotlinx.serialization.json.jsonObject +import kotlinx.serialization.json.jsonPrimitive /** * Runs the compression pipeline (docs/plans/correx-compression-pipeline.md) in fixed pipeline @@ -80,13 +83,21 @@ class DefaultContextPackBuilder( // reorder freely and PromptRenderer can restore true turn order from the ordinal. val stamped = entries.mapIndexed { index, entry -> entry.copy(ordinal = index) } + // A stalled stage re-issuing the exact same call otherwise burns budget on N identical + // (tool, arguments) pairs that all say the same thing — collapse those to the most recent + // occurrence. Fingerprint is (name, arguments) verbatim, so a call whose arguments genuinely + // differ (different start_line/end_line, different path) is a distinct read and untouched. + // Must run before FORMAT_COMPRESS below: that stage reencodes assistantToolCall content from + // raw ToolCallRequest JSON into a dotted-line format the fingerprint parser can't read. + val deduped = dedupeRepeatedToolCalls(stamped) + // Stage 1 (FORMAT_COMPRESS): lossless json→dotted-line compaction of structured entries. val formatted = if (policy.enabled(CompressionStage.FORMAT_COMPRESS)) { - stamped.map { entry -> + deduped.map { entry -> if (classifier.classify(entry) == ContextClass.STRUCTURED) reencode(entry, formatCompressor.compress(entry.content)) else entry } - } else stamped + } else deduped // Stage 3 (TOKEN_PRUNE): prune freeform prose, preserving protected spans. When TIER_SPLIT // is on, the newest TIER0_TURNS freeform turns are left full-fidelity (tier 0). @@ -151,7 +162,7 @@ class DefaultContextPackBuilder( result } - val retained = (pinned + compressed).sortedBy { it.ordinal } + val retained = reconcileToolPairs((pinned + compressed).sortedBy { it.ordinal }) val layers = retained.groupBy { it.layer } val droppedCount = entries.size - retained.count { it.sourceType != "factSheet" } val truncatedLayers = if (droppedCount > 0) { @@ -235,6 +246,65 @@ class DefaultContextPackBuilder( return rest + ranked } + /** + * Collapses assistantToolCall/toolResult pairs that repeat an earlier call's exact (name, + * arguments) — keeping only the most recent occurrence — so a stalled stage repeating the same + * read doesn't show the model N copies of an identical question-and-answer. Content is the + * `ToolCallRequest` JSON built in SessionOrchestrator; parsed generically here (not via the + * `core:inference` type) to avoid a cross-core dependency. Calls whose arguments differ (e.g. + * different start_line/end_line, a different path) fingerprint differently and are untouched — + * that's genuine incremental progress, not repetition. + */ + private fun dedupeRepeatedToolCalls(entries: List): List { + val calls = entries.filter { it.sourceType == "assistantToolCall" } + val callsById = calls.associateBy { it.sourceId } + val fingerprints = calls.mapNotNull { entry -> + toolCallFingerprint(entry.content)?.let { entry.sourceId to it } + } + val dropIds = fingerprints.groupBy({ it.second }, { it.first }) + .values + .filter { it.size > 1 } + .flatMapTo(mutableSetOf()) { sourceIds -> + val keepId = sourceIds.maxWith(compareBy( + { id: String -> if (callsById.getValue(id).reasoning != null) 1 else 0 }, + { id: String -> callsById.getValue(id).ordinal } + )) + sourceIds.filterNot { it == keepId } + } + if (dropIds.isEmpty()) return entries + return entries.filterNot { + (it.sourceType == "assistantToolCall" || it.sourceType == "toolResult") && it.sourceId in dropIds + } + } + + private fun toolCallFingerprint(content: String): String? = runCatching { + val obj = Json.parseToJsonElement(content).jsonObject + val fn = obj["function"]?.jsonObject ?: return null + val name = fn["name"]?.jsonPrimitive?.content ?: return null + val arguments = fn["arguments"]?.jsonPrimitive?.content ?: "" + "$name:$arguments" + }.getOrNull() + + /** + * A completed tool call is represented as two independently-IDed, independently-compressed + * entries — assistantToolCall + toolResult — linked only by [ContextEntry.sourceId]. Because + * `compressed` groups and trims by sourceType (see the `grouped` fold above), one side of a + * pair can survive its group's budget while the other is trimmed away. An orphaned + * assistantToolCall with no matching result renders as "I called this tool" with nothing after + * it — indistinguishable, to the model, from the call silently failing — which drove a live + * read-loop repro (2026-07-08: discovery stage re-issued the same file_read repeatedly once its + * result vanished from two consecutive rounds). Drop whichever side lost its pair so every + * tool-call entry that survives has its result, and vice versa. + */ + private fun reconcileToolPairs(entries: List): List { + val callIds = entries.filter { it.sourceType == "assistantToolCall" }.mapTo(mutableSetOf()) { it.sourceId } + val resultIds = entries.filter { it.sourceType == "toolResult" }.mapTo(mutableSetOf()) { it.sourceId } + return entries.filterNot { + (it.sourceType == "assistantToolCall" && it.sourceId !in resultIds) || + (it.sourceType == "toolResult" && it.sourceId !in callIds) + } + } + private fun reencode(entry: ContextEntry, content: String): ContextEntry = if (content == entry.content) entry else entry.copy(content = content, tokenEstimate = estimateTokens(content)) diff --git a/core/context/src/main/kotlin/com/correx/core/context/model/ContextEntry.kt b/core/context/src/main/kotlin/com/correx/core/context/model/ContextEntry.kt index f689868c..6883d6bf 100644 --- a/core/context/src/main/kotlin/com/correx/core/context/model/ContextEntry.kt +++ b/core/context/src/main/kotlin/com/correx/core/context/model/ContextEntry.kt @@ -32,4 +32,10 @@ data class ContextEntry( // Default OPTIONAL keeps prior behavior for callers that don't stamp buckets (router chat); // the orchestrator stamps REQUIRED on task/constraint/input/failure entries. val bucket: ContextBucket = ContextBucket.OPTIONAL, + // Chain-of-thought that preceded this ASSISTANT turn (assistantToolCall entries only), as + // captured on InferenceResponse.reasoning. Carried as a side channel, not JSON-encoded into + // `content`, so FORMAT_COMPRESS's dotted-line flatten of the tool-call JSON never touches it. + // Threaded back into the request as ChatMessage.reasoning so models that expect their own + // prior reasoning ahead of a tool-call turn (e.g. Gemma) don't degenerate into re-deriving it. + val reasoning: String? = null, ) diff --git a/core/events/src/main/kotlin/com/correx/core/events/events/RepoKnowledgeEvents.kt b/core/events/src/main/kotlin/com/correx/core/events/events/RepoKnowledgeEvents.kt index b4fbc258..ffb2587f 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/events/RepoKnowledgeEvents.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/events/RepoKnowledgeEvents.kt @@ -25,3 +25,18 @@ data class RepoKnowledgeRetrievedEvent( val query: String, val hits: List, ) : EventPayload + +/** + * Candidates that scored above the L3 index's own top-k cutoff but below the retriever's + * minimum-similarity floor, so were never rendered to the agent as "relevant." Without this, + * a hit below the floor is indistinguishable from one that was never a candidate — recorded + * per invariant #9 since the embedding-similarity score is itself an environment observation. + */ +@Serializable +@SerialName("RepoKnowledgeHitsFiltered") +data class RepoKnowledgeHitsFilteredEvent( + val sessionId: SessionId, + val query: String, + val threshold: Float, + val dropped: List, +) : EventPayload diff --git a/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt b/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt index d64e6e4a..5f282021 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt @@ -57,6 +57,7 @@ import com.correx.core.events.events.ModelUnloadedEvent import com.correx.core.events.events.OrchestrationPausedEvent import com.correx.core.events.events.OrchestrationResumedEvent import com.correx.core.events.events.RefinementIterationEvent +import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent import com.correx.core.events.events.RepoKnowledgeRetrievedEvent import com.correx.core.events.events.RepoMapComputedEvent import com.correx.core.events.events.RetryAttemptedEvent @@ -154,6 +155,7 @@ val eventModule = SerializersModule { subclass(RepoMapComputedEvent::class) subclass(WorkspaceStateObservedEvent::class) subclass(RepoKnowledgeRetrievedEvent::class) + subclass(RepoKnowledgeHitsFilteredEvent::class) subclass(BriefGroundingCheckedEvent::class) subclass(BriefEchoMismatchEvent::class) subclass(StaticAnalysisCompletedEvent::class) diff --git a/core/inference/src/main/kotlin/com/correx/core/inference/PromptRenderer.kt b/core/inference/src/main/kotlin/com/correx/core/inference/PromptRenderer.kt index a170fb87..09aef2e7 100644 --- a/core/inference/src/main/kotlin/com/correx/core/inference/PromptRenderer.kt +++ b/core/inference/src/main/kotlin/com/correx/core/inference/PromptRenderer.kt @@ -10,6 +10,14 @@ import kotlinx.serialization.Serializable data class ChatMessage( val role: String, val content: String, + // Carries ContextEntry.reasoning forward for assistant tool-call turns — see that field's + // doc for why this exists (Gemma-family looping when its own prior reasoning is missing). + val reasoning: String? = null, + // Optional tool-call-id that links a role:tool response to the assistant's tool_calls entry + // (sourceId in the event store). Required by the Gemma4 Jinja template to render tool + // response blocks with the correct function name; without it every tool response is + // rendered as "unknown" and the thought/reasoning channel logic can break. + val toolCallId: String? = null, ) object PromptRenderer { @@ -61,5 +69,7 @@ object PromptRenderer { EntryRole.USER -> "user" }, content = content, + reasoning = reasoning, + toolCallId = if (role == EntryRole.TOOL) sourceId else null, ) } diff --git a/core/journal/src/main/kotlin/com/correx/core/journal/DecisionJournalRenderer.kt b/core/journal/src/main/kotlin/com/correx/core/journal/DecisionJournalRenderer.kt index 6c6189f7..50fdc105 100644 --- a/core/journal/src/main/kotlin/com/correx/core/journal/DecisionJournalRenderer.kt +++ b/core/journal/src/main/kotlin/com/correx/core/journal/DecisionJournalRenderer.kt @@ -7,7 +7,7 @@ import com.correx.core.journal.model.DecisionRecord class DecisionJournalRenderer(private val keepLast: Int = 40) { fun render(state: DecisionJournalState, summaryText: String? = null): String { if (state.records.isEmpty() && summaryText == null) return "" - val visible = foldResolvedRetries(state.records) + val visible = foldResolvedRetries(state.records).filter { it.kind in AGENT_RELEVANT_KINDS } return buildString { appendLine("## Decision history (chronological — ground truth)") if (summaryText != null) { @@ -16,7 +16,7 @@ class DecisionJournalRenderer(private val keepLast: Int = 40) { appendLine("+${state.lowSalienceOmittedCount} routine transitions/retries omitted") } } - visible.takeLast(keepLast).forEach { appendLine("- [${it.kind}] ${it.summary}") } + visible.takeLast(keepLast).forEach { appendLine("- ${it.summary}") } }.trimEnd() } @@ -46,4 +46,15 @@ class DecisionJournalRenderer(private val keepLast: Int = 40) { record.copy(summary = "$stage: $count retr${if (count == 1) "y" else "ies"}, resolved — stage completed") } } + + companion object { + /** + * What actually shapes an agent's next decision: the goal, human steering, and failures + * worth not repeating. APPROVAL/ARTIFACT/TRANSITION are bookkeeping for the audit trail — + * agents don't act on "Approval APPROVED (tier T2)" or "Advanced x → y", so they're + * dropped from the render (still present in [DecisionJournalState.records] for the journal). + */ + private val AGENT_RELEVANT_KINDS = + setOf(DecisionKind.INTENT, DecisionKind.STEERING, DecisionKind.PREEMPT, DecisionKind.FAILURE, DecisionKind.RETRY) + } } diff --git a/core/journal/src/test/kotlin/com/correx/core/journal/DecisionJournalRendererTest.kt b/core/journal/src/test/kotlin/com/correx/core/journal/DecisionJournalRendererTest.kt index 3a5c1261..70c1220b 100644 --- a/core/journal/src/test/kotlin/com/correx/core/journal/DecisionJournalRendererTest.kt +++ b/core/journal/src/test/kotlin/com/correx/core/journal/DecisionJournalRendererTest.kt @@ -27,11 +27,28 @@ class DecisionJournalRendererTest { records = listOf(record(1), record(2)), ) val result = renderer.render(state) - assertTrue(result.contains("- [INTENT] summary-1")) - assertTrue(result.contains("- [INTENT] summary-2")) + assertTrue(result.contains("- summary-1")) + assertTrue(result.contains("- summary-2")) assertFalse(result.contains("omitted")) } + @Test + fun `approval, artifact and transition records are excluded from the render`() { + val state = DecisionJournalState( + records = listOf( + record(1, DecisionKind.APPROVAL, "Approval APPROVED (tier T2)"), + record(2, DecisionKind.ARTIFACT, "Validated artifact x at stage y"), + record(3, DecisionKind.TRANSITION, "Advanced a → b (t1)").copy(stageId = "b", fromStageId = "a"), + record(4, DecisionKind.FAILURE, "Stage b failed: boom"), + ), + ) + val result = renderer.render(state) + assertFalse(result.contains("Approval APPROVED")) + assertFalse(result.contains("Validated artifact")) + assertFalse(result.contains("Advanced a")) + assertTrue(result.contains("- Stage b failed: boom")) + } + @Test fun `summaryText appears before records`() { val state = DecisionJournalState( @@ -42,7 +59,7 @@ class DecisionJournalRendererTest { val lines = result.lines() val headerIdx = lines.indexOfFirst { it.startsWith("## Decision history") } val summaryIdx = lines.indexOfFirst { it == "Compaction summary here" } - val recordIdx = lines.indexOfFirst { it.startsWith("- [INTENT]") } + val recordIdx = lines.indexOfFirst { it.startsWith("- summary") } assertTrue(headerIdx >= 0) assertTrue(summaryIdx > headerIdx) assertTrue(recordIdx > summaryIdx) @@ -75,7 +92,7 @@ class DecisionJournalRendererTest { assertTrue(result.contains("## Decision history")) assertTrue(result.contains("Compacted summary")) assertTrue(result.contains("+2 routine transitions/retries omitted")) - assertFalse(result.contains("- [")) + assertFalse(result.contains("- ")) } @Test @@ -90,8 +107,8 @@ class DecisionJournalRendererTest { ) val result = renderer.render(state) assertFalse(result.contains("did not produce"), "resolved retry detail must not render") - assertTrue(result.contains("- [RETRY] scaffold: 2 retries, resolved — stage completed")) - assertTrue(result.contains("- [TRANSITION] Advanced scaffold → hook")) + assertTrue(result.contains("- scaffold: 2 retries, resolved — stage completed")) + assertFalse(result.contains("Advanced scaffold"), "TRANSITION records are audit-only, not agent-relevant") } @Test @@ -102,7 +119,7 @@ class DecisionJournalRendererTest { ), ) val result = renderer.render(state) - assertTrue(result.contains("- [RETRY] Retry 1/3 of hook: did not produce")) + assertTrue(result.contains("- Retry 1/3 of hook: did not produce")) } @Test @@ -115,6 +132,6 @@ class DecisionJournalRendererTest { ), ) val result = renderer.render(state) - assertTrue(result.contains("- [RETRY] Retry 1/3 of scaffold: regression")) + assertTrue(result.contains("- Retry 1/3 of scaffold: regression")) } } diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt index c1704014..da7727af 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt @@ -62,6 +62,7 @@ import com.correx.core.events.events.EventMetadata import com.correx.core.events.events.EventPayload import com.correx.core.events.events.InferenceCompletedEvent import com.correx.core.events.events.RepoKnowledgeHit +import com.correx.core.events.events.InitialIntentEvent import com.correx.core.events.events.RepoKnowledgeRetrievedEvent import com.correx.core.events.events.RepoMapComputedEvent import com.correx.core.events.events.InferenceFailedEvent @@ -582,6 +583,11 @@ abstract class SessionOrchestrator( var toolRounds = 0 var consecutiveReadOnlyRounds = 0 var consecutiveRejectedRounds = 0 + // Fingerprints ("tool:arguments") of every read-only call made during the current + // no-write streak. A round only counts against the read-loop breaker if it repeats calls + // already seen — a stage reading N distinct files before writing is legitimate context + // gathering, not a loop, and shouldn't trip the same counter as re-reading the same file. + val seenReadFingerprints = mutableSetOf() // Set when the model produces its artifact via the emit_artifact tool instead of a final // JSON message; overrides the post-loop capture of the (then-empty) assistant text. var llmArtifactOverride: String? = null @@ -697,7 +703,8 @@ abstract class SessionOrchestrator( } val toolEntries = dispatchToolCalls(sessionId, stageId, inferenceResult.response.toolCalls, stageConfig, effectives, - approvalModeFor(session.state.boundProfile?.approvalMode)) + approvalModeFor(session.state.boundProfile?.approvalMode), + inferenceResult.response.reasoning) val fatalEntry = toolEntries.firstOrNull { it.content.startsWith("FATAL:") } if (fatalEntry != null) { emitProcessResultEvents(sessionId, stageId, stageConfig) @@ -718,14 +725,23 @@ abstract class SessionOrchestrator( if (owesFileWrite() && inferenceResult.response.toolCalls.none { it.function.name in WRITE_TOOL_NAMES } ) { - consecutiveReadOnlyRounds++ - if (consecutiveReadOnlyRounds >= READ_LOOP_NUDGE_THRESHOLD) { + val roundFingerprints = inferenceResult.response.toolCalls + .map { "${it.function.name}:${it.function.arguments}" } + val madeProgress = roundFingerprints.any { it !in seenReadFingerprints } + seenReadFingerprints += roundFingerprints + if (madeProgress) { consecutiveReadOnlyRounds = 0 - inferenceResult = pushBack(readLoopNudge, forceWriteOnly = true) - continue + } else { + consecutiveReadOnlyRounds++ + if (consecutiveReadOnlyRounds >= READ_LOOP_NUDGE_THRESHOLD) { + consecutiveReadOnlyRounds = 0 + inferenceResult = pushBack(readLoopNudge, forceWriteOnly = true) + continue + } } } else { consecutiveReadOnlyRounds = 0 + seenReadFingerprints.clear() } // Rejection-loop breaker (stage-type agnostic): every tool result this round was a // rejection (plane-2 BLOCKED or a stage/policy ERROR) and nothing succeeded. A stage that @@ -1028,10 +1044,14 @@ abstract class SessionOrchestrator( // On a recoverable tool failure, append the tool's argument schema to the error fed back to the // model so it can self-correct its next attempt (it already sees its own malformed call in the // assistant entry above) rather than repeating the same mistake. Kept compact to bound context. + // NOTE: the "corrected arguments" phrasing was removed because it actively drives loops: a model + // that called list_dir frontend/ and got "directory not found" interprets "re-issue with corrected + // arguments" as "retry the same call" rather than "try a different path" — the arguments were + // valid, the path simply doesn't exist. The schema reference stays so malformed calls can still + // self-correct; the model must infer the tactic from the tool's error text. private fun toolArgsHint(tool: Tool?): String = tool?.let { - "\nThe '${it.name}' tool requires arguments matching this JSON schema — re-issue the " + - "call with corrected arguments: ${it.parametersSchema}" + "\nAccepted parameters for '${it.name}': ${it.parametersSchema}" }.orEmpty() private suspend fun dispatchToolCalls( @@ -1041,11 +1061,18 @@ abstract class SessionOrchestrator( stageConfig: StageConfig, effectives: RunEffectives, approvalMode: ApprovalMode, + // The reasoning that preceded this whole batch of tool calls (they all came from one + // inference response). Attached only to the first call's assistant entry — one thought + // block per model turn, matching how the response actually arrived. + reasoning: String? = null, ): List { val executor = effectives.executor ?: return emptyList() val processResultSlots = stageConfig.produces.filter { it.kind.id == "process_result" } val fileWrittenSlots = stageConfig.produces.filter { it.kind.id == "file_written" } - return toolCalls.flatMap { toolCall -> + return toolCalls.withIndex().flatMap { (toolCallIndex, toolCall) -> + // Only the first call's assistant entry carries the reasoning — the whole batch came + // from one model turn/one thought block, so later calls in the same batch get none. + val toolCallReasoning = reasoning.takeIf { toolCallIndex == 0 } val invocationId = ToolInvocationId(UUID.randomUUID().toString()) val parameters = parseToolArguments(toolCall.function.arguments) val request = ToolRequest( @@ -1099,6 +1126,7 @@ abstract class SessionOrchestrator( content = Json.encodeToString(ToolCallRequest.serializer(), toolCall), tokenEstimate = estimateTokens(toolCall.function.arguments), role = EntryRole.ASSISTANT, + reasoning = toolCallReasoning, ), ContextEntry( id = ContextEntryId(UUID.randomUUID().toString()), @@ -1155,6 +1183,7 @@ abstract class SessionOrchestrator( content = Json.encodeToString(ToolCallRequest.serializer(), toolCall), tokenEstimate = estimateTokens(toolCall.function.arguments), role = EntryRole.ASSISTANT, + reasoning = toolCallReasoning, ), ContextEntry( id = ContextEntryId(UUID.randomUUID().toString()), @@ -1223,6 +1252,7 @@ abstract class SessionOrchestrator( content = Json.encodeToString(ToolCallRequest.serializer(), toolCall), tokenEstimate = estimateTokens(toolCall.function.arguments), role = EntryRole.ASSISTANT, + reasoning = toolCallReasoning, ), ContextEntry( id = ContextEntryId(UUID.randomUUID().toString()), @@ -1282,6 +1312,7 @@ abstract class SessionOrchestrator( content = Json.encodeToString(ToolCallRequest.serializer(), toolCall), tokenEstimate = estimateTokens(toolCall.function.arguments), role = EntryRole.ASSISTANT, + reasoning = toolCallReasoning, ), ContextEntry( id = ContextEntryId(UUID.randomUUID().toString()), @@ -1353,6 +1384,7 @@ abstract class SessionOrchestrator( content = Json.encodeToString(ToolCallRequest.serializer(), toolCall), tokenEstimate = estimateTokens(toolCall.function.arguments), role = EntryRole.ASSISTANT, + reasoning = toolCallReasoning, ) val resultContent = when (result) { is ToolResult.Success -> @@ -1702,13 +1734,47 @@ abstract class SessionOrchestrator( return closestPathByFilename(guessed, map.entries.map { it.path }) } + /** + * Resolves the actual prompt text a stage will see, for use as the repo-knowledge query. + * `metadata["prompt"]` holds a *path* to the prompt file (resolved into context separately, + * around line 431) — using that path string itself as the embedding query means the query the + * retriever compares against is a filesystem path, not the stage's real instructions, so + * similarity scores never reflect genuine relevance (2026-07-08 finding: no file scored a + * clear match, everything clustered in the embedder's baseline noise band for short strings). + */ + private fun resolveStagePromptText(stageId: StageId, stageConfig: StageConfig): String { + stageConfig.metadata["promptInline"]?.trim()?.takeIf { it.isNotBlank() }?.let { return it } + stageConfig.metadata["prompt"]?.let { path -> + runCatching { promptResolver.resolve(path) }.getOrNull() + ?.trim()?.takeIf { it.isNotBlank() } + ?.let { return it } + } + return stageId.value + } + + /** + * The role's own prompt text (see [resolveStagePromptText]) is identical across every session + * that runs the stage — it carries zero session-specific signal, so similarity against it can + * only ever reflect generic lexical overlap with the role's boilerplate ("file_read", "ls", + * "grep"...), never what the user actually asked for. The one thing that IS session-specific + * is the initial user intent, so it's prepended here as the real anchor for the query (2026-07-08 + * finding: without it, the query matched a tool script's own symbol names over the files the + * user's request actually named). + */ + private fun repoKnowledgeQuery(sessionId: SessionId, stageId: StageId, stageConfig: StageConfig): String { + val intent = eventStore.read(sessionId) + .mapNotNull { it.payload as? InitialIntentEvent } + .firstOrNull()?.intent?.trim() + val stagePrompt = resolveStagePromptText(stageId, stageConfig) + return if (intent.isNullOrBlank()) stagePrompt else "$intent\n\n$stagePrompt" + } + private suspend fun buildContextualRepoEntries( sessionId: SessionId, stageId: StageId, stageConfig: StageConfig, ): List { - val stagePrompt = (stageConfig.metadata["promptInline"] ?: stageConfig.metadata["prompt"] ?: stageId.value) - .trim().ifBlank { stageId.value } + val stagePrompt = repoKnowledgeQuery(sessionId, stageId, stageConfig) val recorded = eventStore.read(sessionId) .mapNotNull { it.payload as? RepoKnowledgeRetrievedEvent } .lastOrNull { it.stageId == stageId } diff --git a/docs/ui/web-ui-design-spec.md b/docs/ui/web-ui-design-spec.md new file mode 100644 index 00000000..d6b268fd --- /dev/null +++ b/docs/ui/web-ui-design-spec.md @@ -0,0 +1,21 @@ +# Web UI Design Specification + +## Aesthetic +- **Theme**: Soft Rounded + Blue Accent. +- **Consistency**: Should align with the TUI layout plan. + +## Layout +- **Sidebar**: Centralized navigation. +- **Main Area**: Dynamic content based on selection. +- **Event Log**: Easily accessible (Source of Truth). + +## Components +- **Cards**: For task summaries, status updates. +- **Badges**: For status (e.g., `CLAIMED`, `PENDING`, `COMPLETED`). +- **Progress Bars**: For multi-step task progress. +- **Service Layer**: Placeholders for `apps/server` endpoints. + +## UX Priorities +- Clear hierarchy. +- Intuitive navigation. +- Avoid clutter (maintain information density). diff --git a/examples/workflows/prompts/architect_freestyle.md b/examples/workflows/prompts/architect_freestyle.md index 9d8021f1..2967b877 100644 --- a/examples/workflows/prompts/architect_freestyle.md +++ b/examples/workflows/prompts/architect_freestyle.md @@ -85,6 +85,27 @@ Emit a JSON object that validates against the `execution_plan` schema: Those fetch remote code and prompt on stdin; the run is unattended (no TTY) so they hang, and `shell` rejects them outright. `shell` in a scaffold stage is for `npm install` / `npm run build` on files that already exist, not for generating the project. +- **Every stage prompt must describe its exact JSON output structure when using a + structured kind.** The model that runs the stage never sees the raw JSON schema — it only sees + the kind's name (e.g. `analysis`, `research_report`, `review_report`) and the prompt you write + here. Therefore each stage prompt must: + - Conclude with a `Call \`emit_artifact\` with a JSON object matching this shape:` line + followed by the literal JSON structure the kind expects (fields, types, required vs + optional). + - When the kind is `file_written` no `emit_artifact` is needed — the stage writes files + directly. For all other kinds, the prompt must spell out the exact JSON schema inline. + - Include at least one concrete example JSON object matching the schema, so the model has + a template to follow. + - For reference, the available JSON schemas and their shapes are: `discovery` + (`{ready: boolean, questions: [{prompt, options?, multiSelect?, header?}]}`), + `analysis` (`{goal, constraints: [{description, severity}], risks: + [{description, severity, mitigation?}], open_questions: [{question, answered?}]}`), + `design` (`{approach, architecture: [{component, responsibility, interfaces}], + plan: [{step, action, files: [path]}]}`), + `impl_plan` (`{overview, steps: [{action, target, details}]}`), + `research_report` (`{summary, findings: [string], sources: [string]}`), + `review_report` (`{verdict, notes}`). Use these shapes verbatim in the + prompt — do not paraphrase or omit fields. **edges** — describe every transition between stages. Rules: - `from` and `to` must each be a declared stage `id` or the literal string `"done"`. diff --git a/examples/workflows/prompts/implementer.md b/examples/workflows/prompts/implementer.md index 54f4b63c..5d7305e4 100644 --- a/examples/workflows/prompts/implementer.md +++ b/examples/workflows/prompts/implementer.md @@ -1,23 +1,30 @@ You are the **Implementer**. -You receive the `impl_plan` artifact (above). Execute it using the tools available -(`file_read`, `file_write`, `file_edit`, and shell). File writes land in the bound workspace. +You receive the claimed task's acceptance criteria (injected above). Execute them using the tools +available (`file_read`, `file_write`, `file_edit`, and shell). File writes are auto-scoped to the +task's `affected_paths`; widen that scope via `propose_scope` when a needed path is outside it. Steps: -1. If the analysis opened or referenced a task, `task_context` to load it and `task_update - action=claim` before you start (`task_create` one only if the work warrants tracking and none - exists). Skip this for a self-contained change. -2. Work through the plan `steps` in order. Read before you edit. +1. `task_context` to load the claimed task and confirm its acceptance criteria. +2. Work toward satisfying every acceptance criterion. Read before you edit. 3. Make the change with `file_write` / `file_edit`. Keep new code consistent with the surrounding style, naming, and patterns. -4. Run the plan's `verification` commands (build/tests) via shell and fix what fails. Do not +4. **If a `file_write`/`file_edit` is BLOCKED** (the tool result starts with `BLOCKED:`), the + intended path is outside the claimed task's `affected_paths`. The kernel may append a + "Did you mean: " suggestion — that is a string-similarity hint, **not** permission to + write there. Never work around a block by writing the file to a sibling path that happens to + be writable: that creates a shadow/duplicate file that breaks builds and confuses the + reviewer. Instead call `propose_scope` to add the intended path (or its enclosing glob) to + the manifest, then retry the write to the *original* path. If `propose_scope` is denied or + unavailable, stop and report the block in your output — do not invent scope. +5. Run the task's verification commands (build/tests) via shell and fix what fails. Do not leave a step in a broken state. -5. When every step is done and verification passes, `task_update action=submit_for_review` on the - task (if any), then call the `stage_complete` tool. +6. When every criterion is met and verification passes, `task_update action=submit_for_review` + on the task, then call `stage_complete`. The decision history above is ground truth. **If the reviewer requested changes** in a prior round, you will see that verdict and its notes above — address those specific points; do not re-do work that was already approved, and do not repeat a rejected approach. -Implement only what the plan calls for (YAGNI). If a step is genuinely blocked or the plan is -wrong, say so clearly rather than inventing scope. +Implement only what the task's acceptance criteria call for (YAGNI). If something is genuinely +blocked or the criteria are wrong, say so clearly rather than inventing scope. diff --git a/examples/workflows/prompts/reviewer.md b/examples/workflows/prompts/reviewer.md index 04d87e18..3f8d471a 100644 --- a/examples/workflows/prompts/reviewer.md +++ b/examples/workflows/prompts/reviewer.md @@ -1,18 +1,21 @@ You are the **Reviewer** — the quality gate. -You receive the `patch` (the implementer's work), the `impl_plan` it was meant to satisfy, and -the `analysis` whose `requirements` are the **acceptance criteria** for this work. +You receive the `patch` (the implementer's work) and the `analysis` artifact (broader context). +First, use `task_context` to load the claimed task and read its `acceptance_criteria` — those +are the implementer's actual brief, not the full analysis. -Review the **diff against those acceptance criteria** — not whole files against personal taste. -The narrow question is: *does this change satisfy each stated requirement, correctly?* A finding -that doesn't map to a requirement, the plan, or a real defect is noise. +Review the **diff against those task acceptance criteria** — not whole files against personal +taste. The narrow question is: *does this change satisfy each stated criterion, correctly?* A +finding that doesn't map to a task criterion, an analysis requirement, or a real defect is noise. +The `analysis.requirements` are supplementary: use them for cross-checking, but the implementer +may be scoped to only a subset of the full analysis. Review against the **current** ground truth, not a stale one: the decision history above includes any user steering and your own prior verdicts. If the user steered the implementer away from the original plan, judge against the steered direction — not the superseded plan. Check: -1. Does the patch satisfy **every** requirement in `analysis.requirements` and step of `impl_plan`? +1. Does the patch satisfy **every** task acceptance criterion? 2. Is it correct, consistent with the codebase's patterns, and free of obvious defects? 3. Did verification (build/tests) actually pass? 4. No unrequested scope, no placeholders, no regressions. @@ -21,16 +24,15 @@ Do **not** spend the review re-finding what deterministic tools already catch: c formatter/detekt/lint violations, and failing tests are reported by the build itself. If verification passed, trust it; if it failed, the failure is already on the record — point to it, don't re-derive it. Your value is the judgment the tools can't give: whether the change actually -meets the requirements. +meets the acceptance criteria. Emit your result as the `review_report` artifact (JSON, schema provided): - `verdict`: exactly `"approved"` or `"changes_requested"`. - `notes`: if `changes_requested`, list the specific, actionable changes needed (one per - line), each tied to the requirement or plan step it violates, so the implementer knows - exactly what to fix. If `approved`, briefly state which requirements the patch satisfies. + line), each tied to the task criterion or requirement it violates, so the implementer knows + exactly what to fix. If `approved`, briefly state which criteria the patch satisfies. -If the work is tracked as a task, reflect your verdict on it (use `task_context` first if you -need its own acceptance criteria): on `approved`, `task_update action=complete`; on +Reflect your verdict on the task: on `approved`, `task_update action=complete`; on `changes_requested`, leave it claimed for the implementer — optionally add a `note` summarising what's needed. diff --git a/examples/workflows/prompts/task_planner.md b/examples/workflows/prompts/task_planner.md index 611884dd..f45a1fce 100644 --- a/examples/workflows/prompts/task_planner.md +++ b/examples/workflows/prompts/task_planner.md @@ -36,6 +36,12 @@ For `task_decompose`, provide: `affected_paths` (workspace-relative globs), and `depends_on` (refs to other tasks in this batch that must finish first). + **`affected_paths` globs must cover subdirectories recursively.** Use `**/*.tsx` instead of + `*.tsx`, `**/*.py` instead of `*.py`, etc. Sweep the actual directory tree with `list_dir` or + `shell find` before setting paths — if the target tree has subdirectories (e.g. + `components/ui/`, `components/layout/`), a non-recursive glob will jail the implementer to + only the top level and block writes to the real file locations. + Keep the graph small. Over-splitting is worse than under-splitting — a session works one task at a time, and each child is a future claim. Aim for 2–5 children unless the request genuinely demands more. diff --git a/infrastructure/inference/llama_cpp/src/main/kotlin/com/correx/infrastructure/inference/llama/cpp/LlamaCppInferenceProvider.kt b/infrastructure/inference/llama_cpp/src/main/kotlin/com/correx/infrastructure/inference/llama/cpp/LlamaCppInferenceProvider.kt index 65bbb22c..f0712812 100644 --- a/infrastructure/inference/llama_cpp/src/main/kotlin/com/correx/infrastructure/inference/llama/cpp/LlamaCppInferenceProvider.kt +++ b/infrastructure/inference/llama_cpp/src/main/kotlin/com/correx/infrastructure/inference/llama/cpp/LlamaCppInferenceProvider.kt @@ -142,9 +142,15 @@ class LlamaCppInferenceProvider( val baseMessages = PromptRenderer.render(request.contextPack) val messages = baseMessages.map { msg -> + val toolCalls = if (msg.role == "assistant") { + parseAssistantToolCalls(msg.content) + } else emptyList() LlamaCppChatMessage( role = msg.role, - content = msg.content, + content = if (toolCalls.isNotEmpty()) null else msg.content, + reasoningContent = msg.reasoning, + toolCalls = toolCalls, + toolCallId = msg.toolCallId, ) } @@ -230,5 +236,40 @@ class LlamaCppInferenceProvider( ProviderHealth.Unavailable("Health check failed: ${e.message}") } + /** + * Parses an assistant message's content back into [ToolCallRequest] objects. + * + * The context builder's FORMAT_COMPRESS stage converts raw ToolCallRequest JSON into a compact + * dotted-line format (id = ..., function.name = ...). If compression is disabled, the content + * is still valid JSON. This handles both forms so the Gemma4 Jinja template sees the native + * `tool_calls` array — which it requires to render `reasoning_content` as a `<|channel>thought` + * block (without `tool_calls` the reasoning field is silently dropped and the model re-enters + * thought mode from scratch, causing the degenerate `<|channel>thought`~16384-token loop). + */ + private fun parseAssistantToolCalls(content: String): List { + // Try JSON parse first (FORMAT_COMPRESS disabled — raw ToolCallRequest JSON). + if (content.startsWith("{\"id\"")) { + runCatching { + return listOf(json.decodeFromString(content)) + } + } + // Try dotted-line format (FORMAT_COMPRESS enabled). + if (!content.contains("function.name")) return emptyList() + val lines = content.split('\n') + var id: String? = null + var name: String? = null + var arguments: String? = null + for (line in lines) { + when { + line.startsWith("id = ") -> id = line.removePrefix("id = ").trim() + line.startsWith("function.name = ") -> name = line.removePrefix("function.name = ").trim() + line.startsWith("function.arguments = ") -> arguments = line.removePrefix("function.arguments = ").trim() + } + } + return if (name != null && arguments != null) { + listOf(ToolCallRequest(id = id, function = ToolCallFunction(name = name, arguments = arguments))) + } else emptyList() + } + override fun capabilities(): Set = descriptor.capabilities } diff --git a/infrastructure/inference/openai_compat/src/main/kotlin/com/correx/infrastructure/inference/openai/OpenAiCompatInferenceProvider.kt b/infrastructure/inference/openai_compat/src/main/kotlin/com/correx/infrastructure/inference/openai/OpenAiCompatInferenceProvider.kt index 4d68202a..34c26d4a 100644 --- a/infrastructure/inference/openai_compat/src/main/kotlin/com/correx/infrastructure/inference/openai/OpenAiCompatInferenceProvider.kt +++ b/infrastructure/inference/openai_compat/src/main/kotlin/com/correx/infrastructure/inference/openai/OpenAiCompatInferenceProvider.kt @@ -88,7 +88,7 @@ class OpenAiCompatInferenceProvider( // tool_calls turn (we don't carry tool_call_ids through the context pack). Fold any // non-standard role into a user turn so the model still sees the content. if (msg.role in STANDARD_ROLES) { - OpenAiChatMessage(role = msg.role, content = msg.content) + OpenAiChatMessage(role = msg.role, content = msg.content, reasoningContent = msg.reasoning) } else { OpenAiChatMessage(role = "user", content = "[${msg.role}] ${msg.content}") } diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt index 46e125bd..009e9374 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt @@ -135,8 +135,18 @@ class FileEditTool( operation == "patch" && !request.parameters.containsKey("patch") -> ValidationResult.Invalid("Missing 'patch' parameter for 'patch' operation.") + // Don't point the model at 'file_read'/'list_dir' here — this rejection is commonly hit + // in exactly the round those tools have been deliberately withheld (the read-loop + // breaker forces a write-only round after repeated reads with no write), so telling it + // to go use tools that aren't in its tool list this round would just compound the stall. + operation == "read" -> + ValidationResult.Invalid( + "'edit_file' has no 'read' operation. It only supports 'append', 'replace', or 'patch' — " + + "proceed with one of those using what you already know about the file.", + ) + operation !in listOf("append", "replace", "patch") -> - ValidationResult.Invalid("Unknown operation: $operation") + ValidationResult.Invalid("Unknown operation: $operation. Expected 'append', 'replace', or 'patch'.") else -> ValidationResult.Valid } diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileWriteTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileWriteTool.kt index a9a7a9f5..e7e78310 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileWriteTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileWriteTool.kt @@ -154,6 +154,7 @@ class FileWriteTool( val originalContent = runCatching { Files.readString(path) }.getOrDefault("") val result = runCatching { + Files.createDirectories(path.parent) AtomicFileWriter.write(path, content.toByteArray(Charsets.UTF_8)) ToolResult.Success( invocationId = request.invocationId, diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt index c306fc35..17394622 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt @@ -97,8 +97,17 @@ class ListDirTool( val recursive = request.parameters["recursive"]?.toString()?.toBoolean() ?: false val root = resolvePath(pathString) when { - !Files.exists(root) -> - ToolResult.Failure(request.invocationId, "Directory not found: $pathString", recoverable = true) + !Files.exists(root) -> { + val suggestions = similarEntries(root) + val msg = buildString { + append("Directory not found: $pathString") + if (suggestions.isNotEmpty()) { + append(". Did you mean ") + append(suggestions.joinToString(", ", postfix = "?")) + } + } + ToolResult.Failure(request.invocationId, msg, recoverable = true) + } !Files.isDirectory(root) -> ToolResult.Failure(request.invocationId, "Not a directory: $pathString", recoverable = true) else -> runCatching { walk(root, recursive, request) }.getOrElse { @@ -109,8 +118,13 @@ class ListDirTool( private fun walk(root: Path, recursive: Boolean, request: ToolRequest): ToolResult { val matcher = GitignoreMatcher(workingDir ?: root, root) + // The caller explicitly targeted `root` — if a rule already covers it (e.g. an entire + // subtree gitignored as scaffold output), don't let that same rule keep pruning everything + // found underneath; other, unrelated rules (node_modules/ inside it, etc.) still apply. + matcher.exemptRulesMatching(root) val out = ArrayList() var truncated = false + var ignoredCount = 0 // Emit one entry; TERMINATE the walk once the cap is hit so a huge tree can't run away. fun emit(path: Path, isDir: Boolean): FileVisitResult { val slash = if (isDir) "/" else "" @@ -121,8 +135,11 @@ class ListDirTool( val maxDepth = if (recursive) Int.MAX_VALUE else 1 Files.walkFileTree(root, emptySet(), maxDepth, object : SimpleFileVisitor() { override fun preVisitDirectory(dir: Path, attrs: BasicFileAttributes): FileVisitResult { - if (dir != root && (dir.name == ".git" || matcher.ignored(dir, isDir = true))) + if (dir != root && dir.name == ".git") return FileVisitResult.SKIP_SUBTREE + if (dir != root && matcher.ignored(dir, isDir = true)) { + ignoredCount++ return FileVisitResult.SKIP_SUBTREE + } matcher.loadFrom(dir) return if (dir == root) FileVisitResult.CONTINUE else emit(dir, isDir = true) } @@ -130,15 +147,83 @@ class ListDirTool( // A directory sitting exactly at maxDepth is delivered here (walkFileTree won't // descend it), so honour its dir-ness for both the ignore check and the trailing '/'. val isDir = attrs.isDirectory - return if (matcher.ignored(file, isDir)) FileVisitResult.CONTINUE else emit(file, isDir) + if (matcher.ignored(file, isDir)) { + ignoredCount++ + return FileVisitResult.CONTINUE + } + return emit(file, isDir) } override fun visitFileFailed(file: Path, exc: java.io.IOException) = FileVisitResult.CONTINUE }) out.sort() - val body = out.joinToString("\n").ifEmpty { "(empty)" } - val note = - if (truncated) "\n… truncated at $MAX_ENTRIES entries; narrow the path or list a sub-directory." else "" - return ToolResult.Success(request.invocationId, output = body + note) + return ToolResult.Success(request.invocationId, output = render(root, out, truncated, ignoredCount)) + } + + /** + * `//` wrapper (matching the shape other coding-agent tools use) instead of + * a bare newline list — the count and explicit `directory` tag let a small model tell "empty + * listing" from "malformed output" and know how many entries to expect. Notes on truncation/ + * gitignore-pruning are surfaced here too, so a shorter-than-expected listing isn't mistaken for + * the real tree. + */ + private fun render(root: Path, entries: List, truncated: Boolean, ignoredCount: Int): String { + val notes = buildList { + if (truncated) add("truncated at $MAX_ENTRIES entries; narrow the path or list a sub-directory") + if (ignoredCount > 0) add("$ignoredCount ${if (ignoredCount == 1) "entry" else "entries"} hidden by .gitignore") + } + return buildString { + appendLine("$root") + appendLine("directory") + appendLine("") + appendLine(entries.joinToString("\n").ifEmpty { "(empty)" }) + if (notes.isNotEmpty()) appendLine(notes.joinToString("; ")) + append("(${entries.size} ${if (entries.size == 1) "entry" else "entries"})") + appendLine() + append("") + } + } + + private fun similarEntries(target: Path): List { + val parent = target.parent ?: return emptyList() + if (!Files.isDirectory(parent)) return emptyList() + val targetName = target.fileName?.toString() ?: return emptyList() + return try { + Files.list(parent).use { stream -> + stream.map { it.fileName.toString() } + .filter { it != targetName && isSimilarName(it, targetName) } + .toList() + .sortedWith(compareByDescending { commonPrefixLen(it, targetName) }.thenBy { it }) + .take(3) + } + } catch (_: Exception) { emptyList() } + } + + private fun isSimilarName(a: String, b: String): Boolean { + val prefix = commonPrefixLen(a, b) + return prefix >= 2 || levenshtein(a, b) <= 3 + } + + private fun commonPrefixLen(a: String, b: String): Int { + var i = 0 + val minLen = minOf(a.length, b.length) + while (i < minLen && a[i].lowercaseChar() == b[i].lowercaseChar()) i++ + return i + } + + private fun levenshtein(a: String, b: String): Int { + val dp = Array(a.length + 1) { IntArray(b.length + 1) } + for (i in 0..a.length) dp[i][0] = i + for (j in 0..b.length) dp[0][j] = j + for (i in 1..a.length) { + for (j in 1..b.length) { + dp[i][j] = minOf( + dp[i - 1][j] + 1, + dp[i][j - 1] + 1, + dp[i - 1][j - 1] + if (a[i - 1].lowercaseChar() == b[i - 1].lowercaseChar()) 0 else 1 + ) + } + } + return dp[a.length][b.length] } private companion object { @@ -176,10 +261,17 @@ internal class GitignoreMatcher(anchor: Path, target: Path) { } } - fun ignored(path: Path, isDir: Boolean): Boolean = rules.any { r -> + fun ignored(path: Path, isDir: Boolean): Boolean = rules.any { matches(it, path, isDir) } + + /** Drop whichever rules currently cover [path] — used so an explicitly-targeted (already + * gitignored) listing root doesn't have that same rule keep pruning its own children. */ + fun exemptRulesMatching(path: Path) { + rules.removeAll { matches(it, path, isDir = true) } + } + + private fun matches(r: Rule, path: Path, isDir: Boolean): Boolean = (!r.dirOnly || isDir) && path.startsWith(r.base) && r.regex.matches(r.base.relativize(path).toString().replace('\\', '/')) - } private fun compile(base: Path, raw: String): Rule { var pat = raw diff --git a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/PlanLinter.kt b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/PlanLinter.kt index 5ce6a176..921a5d81 100644 --- a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/PlanLinter.kt +++ b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/PlanLinter.kt @@ -41,7 +41,7 @@ object PlanLinter { fun lint(graph: WorkflowGraph, seeds: Set = seedArtifacts): PlanLintResult = PlanLintResult( - hardFailures = unproducedNeeds(graph, seeds) + trapStates(graph), + hardFailures = unproducedNeeds(graph, seeds) + trapStates(graph) + unreferencedPromptArtifacts(graph, seeds), softFindings = stageCount(graph) + fanOut(graph) + emptyBriefs(graph) + duplicateBriefs(graph), ) @@ -89,6 +89,35 @@ object PlanLinter { } } + /** + * H3: a stage's inline prompt mentions an artifact ID that is produced by the plan or + * provided by a seed, but the stage does not declare it in its `needs`. The stage will + * not have access to that artifact at runtime, so any reference in the prompt is a dangling + * assumption that will confuse or stall the model — force the architect to either add the + * artifact to `needs` or reword the prompt. + */ + private fun unreferencedPromptArtifacts(graph: WorkflowGraph, seeds: Set): List { + val produced = graph.stages.values.flatMap { it.produces }.map { it.name.value }.toSet() + seeds + return graph.stages.entries.flatMap { (id, stage) -> + val prompt = briefOf(stage) + val needIds = stage.needs.map { it.value }.toSet() + val ownProduces = stage.produces.map { it.name.value }.toSet() + produced.filter { prodId -> + prompt.isNotBlank() && + prodId.length >= 2 && // skip single-char IDs (false-positive risk with substring matching) + Regex("\\b${Regex.escape(prodId)}\\b").containsMatchIn(prompt) && + prodId !in needIds && + prodId !in ownProduces + }.map { prodId -> + PlanLintFinding( + code = "unreferenced_prompt_artifact", + stageId = id.value, + detail = "prompt mentions '$prodId' but it is not declared in needs", + ) + } + } + } + /** S1: more stages than the ceiling. */ private fun stageCount(graph: WorkflowGraph): List = if (graph.stages.size > STAGE_CEILING) { diff --git a/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/PlanLinterTest.kt b/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/PlanLinterTest.kt index f0c19451..df556f20 100644 --- a/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/PlanLinterTest.kt +++ b/infrastructure/workflow/src/test/kotlin/com/correx/infrastructure/workflow/PlanLinterTest.kt @@ -92,6 +92,37 @@ class PlanLinterTest { assertTrue(result.passed, "analysis is a planning-phase seed, not an unproduced need") } + @Test + fun `a stage mentioning its own produced artifact in the prompt is not a failure`() { + val g = graph( + stages = mapOf( + "a" to stage(produces = listOf("plan"), prompt = "write the plan artifact"), + "b" to stage(needs = listOf("plan"), prompt = "build"), + ), + edges = listOf("a" to "b", "b" to "done"), + start = "a", + ) + val result = PlanLinter.lint(g) + assertTrue(result.passed, "expected pass, hard=${result.hardFailures}") + } + + @Test + fun `a prompt mentioning another stage's artifact not in needs is a hard failure`() { + val g = graph( + stages = mapOf( + "a" to stage(produces = listOf("plan"), prompt = "write the plan artifact"), + "b" to stage(produces = listOf("report"), prompt = "use plan to build"), + ), + edges = listOf("a" to "b", "b" to "done"), + start = "a", + ) + val result = PlanLinter.lint(g) + assertFalse(result.passed) + val h = result.hardFailures.single { it.code == "unreferenced_prompt_artifact" } + assertEquals("b", h.stageId) + assertTrue(h.detail.contains("plan")) + } + @Test fun `a cycle with no exit is a trap-state hard failure`() { val g = graph( diff --git a/testing/deterministic/src/test/kotlin/DefaultContextPackBuilderTest.kt b/testing/deterministic/src/test/kotlin/DefaultContextPackBuilderTest.kt index 497cb24b..de3d4449 100644 --- a/testing/deterministic/src/test/kotlin/DefaultContextPackBuilderTest.kt +++ b/testing/deterministic/src/test/kotlin/DefaultContextPackBuilderTest.kt @@ -134,10 +134,10 @@ class DefaultContextPackBuilderTest { // layer reads call, result, call, result — not all calls then all results. val entries = listOf( entry("task", ContextLayer.L1, 10).copy(sourceType = "agentPrompt"), - entry("call1", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall"), - entry("result1", ContextLayer.L2, 10).copy(sourceType = "toolResult"), - entry("call2", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall"), - entry("result2", ContextLayer.L2, 10).copy(sourceType = "toolResult"), + entry("call1", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall", sourceId = "tool1"), + entry("result1", ContextLayer.L2, 10).copy(sourceType = "toolResult", sourceId = "tool1"), + entry("call2", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall", sourceId = "tool2"), + entry("result2", ContextLayer.L2, 10).copy(sourceType = "toolResult", sourceId = "tool2"), ) val pack = builder.build(packId, sessionId, stageId, entries, TokenBudget(limit = 4000)) val l2Ids = pack.layers[ContextLayer.L2]!!.map { it.id.value } diff --git a/testing/deterministic/src/test/kotlin/PromptRendererOrderingTest.kt b/testing/deterministic/src/test/kotlin/PromptRendererOrderingTest.kt index e0cbf1a0..79b44ba8 100644 --- a/testing/deterministic/src/test/kotlin/PromptRendererOrderingTest.kt +++ b/testing/deterministic/src/test/kotlin/PromptRendererOrderingTest.kt @@ -18,15 +18,16 @@ class PromptRendererOrderingTest { private val sessionId = SessionId("s1") private val stageId = StageId("stage-1") - private fun entry(id: String, layer: ContextLayer, role: EntryRole, sourceType: String) = ContextEntry( - id = ContextEntryId(id), - layer = layer, - content = id, - sourceType = sourceType, - sourceId = id, - tokenEstimate = 10, - role = role, - ) + private fun entry(id: String, layer: ContextLayer, role: EntryRole, sourceType: String, sourceId: String = id) = + ContextEntry( + id = ContextEntryId(id), + layer = layer, + content = id, + sourceType = sourceType, + sourceId = sourceId, + tokenEstimate = 10, + role = role, + ) @Test fun `stage tool loop renders chronologically with the task before tool turns`() = kotlinx.coroutines.runBlocking { @@ -36,10 +37,10 @@ class PromptRendererOrderingTest { val entries = listOf( entry("sys", ContextLayer.L0, EntryRole.SYSTEM, "systemPrompt"), entry("task", ContextLayer.L1, EntryRole.USER, "agentPrompt"), - entry("call1", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall"), - entry("result1", ContextLayer.L2, EntryRole.TOOL, "toolResult"), - entry("call2", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall"), - entry("result2", ContextLayer.L2, EntryRole.TOOL, "toolResult"), + entry("call1", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall", sourceId = "tool1"), + entry("result1", ContextLayer.L2, EntryRole.TOOL, "toolResult", sourceId = "tool1"), + entry("call2", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall", sourceId = "tool2"), + entry("result2", ContextLayer.L2, EntryRole.TOOL, "toolResult", sourceId = "tool2"), ) val pack = builder.build(ContextPackId("p"), sessionId, stageId, entries, TokenBudget(limit = 4000))