feat(context): reasoning-thread continuity, L3 doc filtering, and gate hardening

Threads reasoning_content across inference calls and journal renders, filters
markdown noise out of L3 repo-knowledge retrieval, adds PlanLinter H3 checks,
and tightens filesystem tool output/dir-listing behavior surfaced by prior
live QA (see project_readloop_campaign memory).
This commit is contained in:
2026-07-10 11:13:43 +04:00
parent f51a8dada4
commit 3d5e05c1fb
32 changed files with 742 additions and 85 deletions
+14
View File
@@ -0,0 +1,14 @@
# Repo-map exclusions — silences generic-similarity noise in the semantic "Relevant files"
# feed without touching CorrexConfig's ignore_globs. One pattern per line, matched by
# RepoMapIndexer.isIgnored: bare name = matches at any depth, path with '/' = subtree prefix.
# Dev/QA tooling, not project business logic.
scripts
# Fixture repo used by QA gate tests — deliberately fake code, not part of correx itself.
qa/sample-repo
# Frontend build config — boilerplate, never what a code-context query is actually after.
frontend/tailwind.config.js
frontend/vite.config.ts
frontend/postcss.config.js
@@ -367,6 +367,7 @@ fun main() {
embedder = embedder,
l3MemoryStore = l3MemoryStore,
repoRoot = workspaceRoot.toString(),
eventStore = eventStore,
)
} else {
null
@@ -1,25 +1,74 @@
package com.correx.apps.server.memory
import com.correx.core.events.events.EventMetadata
import com.correx.core.events.events.NewEvent
import com.correx.core.events.events.RepoKnowledgeHit
import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent
import com.correx.core.events.stores.EventStore
import com.correx.core.events.types.EventId
import com.correx.core.events.types.SessionId
import com.correx.core.inference.Embedder
import com.correx.core.kernel.orchestration.RepoKnowledgeRetriever
import com.correx.core.talkie.l3.L3MemoryStore
import com.correx.core.talkie.l3.L3Query
import kotlinx.coroutines.CancellationException
import kotlinx.datetime.Clock
import org.slf4j.LoggerFactory
import java.util.UUID
private val log = LoggerFactory.getLogger(L3RepoKnowledgeRetriever::class.java)
private const val RETRIEVAL_OVERSAMPLE_FACTOR = 4
// Real embedder scores for genuinely relevant repo-map hits cluster 0.68-0.80 (2026-07-03
// live grounding); below this floor a hit is just the nearest thing in a small corpus, not
// actually related to the query — rendering it as "relevant" is the same poisoning failure
// mode as the markdown-in-L3 bug, just via low-signal cosine similarity instead of topic drift.
private const val MIN_SIMILARITY_SCORE = 0.5f
class L3RepoKnowledgeRetriever(
private val embedder: Embedder,
private val l3MemoryStore: L3MemoryStore,
private val repoRoot: String,
private val eventStore: EventStore? = null,
) : RepoKnowledgeRetriever {
override suspend fun retrieve(sessionId: SessionId, query: String, k: Int): List<RepoKnowledgeHit> {
val vector = embedder.embed(query)
return l3MemoryStore.query(L3Query(vector = vector, k = k * RETRIEVAL_OVERSAMPLE_FACTOR))
val candidates = l3MemoryStore.query(L3Query(vector = vector, k = k * RETRIEVAL_OVERSAMPLE_FACTOR))
// Trailing ':' so "/repo" does not also match "/repo2" turnIds (prefix collision).
.filter { it.entry.turnId.startsWith("repomap:$repoRoot:") }
.take(k)
.map { RepoKnowledgeHit(path = it.entry.text.substringBefore(":"), text = it.entry.text, score = it.score) }
val (kept, dropped) = candidates.partition { it.score >= MIN_SIMILARITY_SCORE }
if (dropped.isNotEmpty()) recordDropped(sessionId, query, dropped.map { it.toHit() })
return kept.take(k).map { it.toHit() }
}
private suspend fun recordDropped(sessionId: SessionId, query: String, dropped: List<RepoKnowledgeHit>) {
val store = eventStore ?: return
runCatching {
store.append(
NewEvent(
metadata = EventMetadata(
eventId = EventId(UUID.randomUUID().toString()),
sessionId = sessionId,
timestamp = Clock.System.now(),
schemaVersion = 1,
causationId = null,
correlationId = null,
),
payload = RepoKnowledgeHitsFilteredEvent(
sessionId = sessionId,
query = query,
threshold = MIN_SIMILARITY_SCORE,
dropped = dropped,
),
),
)
}.exceptionOrNull()?.let { e ->
if (e is CancellationException) throw e
log.warn("failed to record filtered repo-knowledge hits for {}: {}", sessionId.value, e.message)
}
}
private fun com.correx.core.talkie.l3.L3Hit.toHit() =
RepoKnowledgeHit(path = entry.text.substringBefore(":"), text = entry.text, score = score)
}
@@ -117,7 +117,14 @@ class ProjectMemoryService(
// Trailing ':' delimiter so a repoRoot prefix can't collide with a longer sibling
// (/repo vs /repo2) under the retriever's startsWith filter.
val tag = if (stateKey != null) "repomap:$repoRoot:$stateKey" else "repomap:$repoRoot:"
entries.forEach { entry ->
// Docs are already surfaced via the "Docs available" catalog built straight from
// RepoMapComputedEvent (SessionOrchestrator, 2026-07-07 doc-injection rework) — embedding
// them into the same L3 namespace as code let generic textual similarity (e.g. "analyst",
// "compression pipeline") outrank the actual code files for a code-focused query, silently
// crowding the semantic "Relevant files" feed with irrelevant markdown (2026-07-08 recurrence
// of the 2026-07-05 context-poisoning root cause, this time via L3RepoKnowledgeRetriever
// rather than the repo-map floor that was fixed then).
entries.filterNot { it.path.endsWith(".md", ignoreCase = true) }.forEach { entry ->
runCatching {
val text = entry.path + if (entry.symbols.isEmpty()) "" else ": ${entry.symbols.joinToString(", ")}"
l3MemoryStore.store(
@@ -32,12 +32,12 @@ class RepoMapIndexer(
) : RepoMapIndexerPort {
override fun index(repoRoot: Path): List<RepoMapEntry> {
if (!repoRoot.isDirectory()) return emptyList()
val ignore = ignoreGlobs.toSet()
val ignore = ignoreGlobs.toSet() + loadCorrexIgnore(repoRoot)
val files = Files.walk(repoRoot, maxDepth).use { stream ->
stream.filter { it.isRegularFile() }
.filter { it.extension in SYMBOL_PATTERNS.keys }
.filter { path -> path.relativeTo(repoRoot).none { it.name in ignore } }
.filter { path -> !isIgnored(path.relativeTo(repoRoot), ignore) }
.toList()
}
if (files.isEmpty()) return emptyList()
@@ -58,6 +58,35 @@ class RepoMapIndexer(
.sortedByDescending { it.score }
}
/**
* A repo-local, gitignore-style opt-out for the repo map: patterns here apply on top of
* `[project] ignore_globs` (CorrexConfig) without needing a config edit, so anyone working in
* the repo can silence a noisy path for themselves. Same matching rules as [isIgnored] — one
* pattern per line, blank lines and `#` comments skipped. Missing file is not an error (most
* repos won't have one).
*/
private fun loadCorrexIgnore(repoRoot: Path): List<String> {
val file = repoRoot.resolve(".correxignore")
if (!file.isRegularFile()) return emptyList()
return runCatching { file.readText() }.getOrNull()
?.lines()
?.map { it.trim() }
?.filter { it.isNotEmpty() && !it.startsWith("#") }
?: emptyList()
}
/**
* An ignore entry matches either a bare path-segment name (any depth, e.g. "node_modules") or a
* relative path prefix (e.g. "examples/workflows/prompts", to exclude just that subtree).
*/
private fun isIgnored(relativePath: Path, ignore: Set<String>): Boolean {
val rel = relativePath.toString().replace('\\', '/')
return ignore.any { pattern ->
if ('/' in pattern) rel == pattern || rel.startsWith("$pattern/")
else relativePath.any { it.name == pattern }
}
}
private fun extractSymbols(path: Path): List<String> {
// Docs are indexed as a single "what is this" descriptor, not a table-of-contents dump: a
// heading flood (40 h1h3 per doc across a large docs/ tree) buries real source and force-feeds
+18
View File
@@ -89,6 +89,22 @@ type EventEntry struct {
Detail string
}
// PlanStageStatus tracks a stage's lifecycle in a locked execution plan.
type PlanStageStatus int
const (
PlanPending PlanStageStatus = iota
PlanRunning
PlanCompleted
PlanFailed
)
// PlanStage is one stage in a locked execution plan, with live execution status.
type PlanStage struct {
ID string
Status PlanStageStatus
}
// ToolStatus mirrors the Kotlin ToolDisplayStatus.
type ToolStatus int
@@ -194,6 +210,8 @@ type Session struct {
Name string
LastEventAt int64
CurrentStage string
PlanGoal string
PlanStages []PlanStage
WorkspaceRoot string // bound cwd, from session.workspace_bound
LastOutput string
LastResponse string
@@ -615,6 +615,8 @@ var nonRenderingEventTypes = map[string]string{
protocol.TypeFileList: "query reply: populates the @ file picker overlay, not the transcript",
protocol.TypeGrantList: "query reply: populates the standing-grants overlay, not the transcript",
protocol.TypeHealthChecks: "query reply: populates the health-checks overlay, not the transcript",
protocol.TypeProjectProfile: "query reply: populates the project profile overlay, not the transcript",
protocol.TypeOperatorProfile: "query reply: populates the operator profile overlay, not the transcript",
}
// TestRenderMatrixCoversEveryEventType is the load-bearing matrix guarantee: it scans
+23
View File
@@ -153,16 +153,34 @@ func (m *Model) applyServer(msg protocol.ServerMessage) {
s.Tools = nil
s.StageTokensUsed = 0
s.addEvent(msg.OccurredAt, "StageStarted", msg.StageID)
for i := range s.PlanStages {
if s.PlanStages[i].ID == msg.StageID {
s.PlanStages[i].Status = PlanRunning
break
}
}
}
case protocol.TypeStageCompleted:
if s := m.session(msg.SessionID); s != nil {
s.CurrentStage = ""
s.addEvent(msg.OccurredAt, "StageCompleted", msg.StageID)
for i := range s.PlanStages {
if s.PlanStages[i].ID == msg.StageID {
s.PlanStages[i].Status = PlanCompleted
break
}
}
}
case protocol.TypeStageFailed:
if s := m.session(msg.SessionID); s != nil {
s.CurrentStage = ""
s.addEvent(msg.OccurredAt, "StageFailed", msg.StageID)
for i := range s.PlanStages {
if s.PlanStages[i].ID == msg.StageID {
s.PlanStages[i].Status = PlanFailed
break
}
}
}
case protocol.TypeInferenceStarted:
if s := m.session(msg.SessionID); s != nil {
@@ -303,6 +321,11 @@ func (m *Model) applyServer(msg protocol.ServerMessage) {
if s := m.session(msg.SessionID); s != nil {
s.addEvent(nowMillis(), "PlanLocked", strings.Join(msg.StageIDs, " → "))
s.LastOutput = "plan locked: " + strings.Join(msg.StageIDs, " → ")
s.PlanGoal = msg.Content
s.PlanStages = make([]PlanStage, 0, len(msg.StageIDs))
for _, id := range msg.StageIDs {
s.PlanStages = append(s.PlanStages, PlanStage{ID: id, Status: PlanPending})
}
}
case protocol.TypeWorkspaceBound:
if s := m.session(msg.SessionID); s != nil {
+51
View File
@@ -790,6 +790,10 @@ func (m Model) buildTranscriptRows(w int) ([]string, []int) {
t := m.theme
msgs := m.routerMessages[m.selectedID]
var rows []string
// Prepend the live execution plan card when the session has locked plan stages.
if planRows := m.planCardLines(); len(planRows) > 0 {
rows = append(rows, planRows...)
}
msgStart := make([]int, len(msgs)) // first row index of each message, for scroll-to-selection
for mi, e := range msgs {
msgStart[mi] = len(rows)
@@ -847,6 +851,53 @@ func (m Model) buildTranscriptRows(w int) ([]string, []int) {
return rows, msgStart
}
// planCardLines renders the live execution plan card for the selected session.
func (m Model) planCardLines() []string {
t := m.theme
s := m.session(m.selectedID)
if s == nil || len(s.PlanStages) == 0 {
return nil
}
goal := strings.TrimSpace(s.PlanGoal)
var out []string
if goal != "" {
out = append(out, t.span(goal, t.P.Faint))
}
stageLabel := "stage"
if len(s.PlanStages) != 1 {
stageLabel = "stages"
}
out = append(out, t.span("Plan \u00b7 "+itoa(len(s.PlanStages))+" "+stageLabel, t.P.Accent))
idW := 0
for _, ps := range s.PlanStages {
if len(ps.ID) > idW {
idW = len(ps.ID)
}
}
for _, ps := range s.PlanStages {
var badge, badgeStyle string
var badgeColor color.Color
switch ps.Status {
case PlanPending:
badge = "\u25cb"
badgeColor = t.P.Faint
case PlanRunning:
badge = "\u25cf"
badgeColor = t.P.Accent
case PlanCompleted:
badge = "\u2713"
badgeColor = t.P.OK
case PlanFailed:
badge = "\u2717"
badgeColor = t.P.Bad
}
badgeStyle = t.span(badge, badgeColor)
rest := t.span(" "+padRaw(ps.ID, idW), t.P.Fg)
out = append(out, " "+badgeStyle+rest)
}
return out
}
func (m Model) eventRows(w, h int) []string {
t := m.theme
header := lipgloss.NewStyle().Foreground(t.P.Accent).Background(t.P.Bg).Bold(true).Render("event stream") +
@@ -94,6 +94,10 @@ data class ProjectConfig(
companion object {
val DEFAULT_IGNORES: List<String> = listOf(
".git", "node_modules", "build", "target", ".gradle", "dist", ".idea", ".opencode",
// Role-prompt templates: loaded by workflow config as LLM system prompts, not project
// docs — indexing them lets a stage `file_read` another role's prompt as if it were
// documentation.
"examples/workflows/prompts",
)
}
}
@@ -26,6 +26,9 @@ import com.correx.core.events.types.ContextEntryId
import com.correx.core.events.types.ContextPackId
import com.correx.core.events.types.SessionId
import com.correx.core.events.types.StageId
import kotlinx.serialization.json.Json
import kotlinx.serialization.json.jsonObject
import kotlinx.serialization.json.jsonPrimitive
/**
* Runs the compression pipeline (docs/plans/correx-compression-pipeline.md) in fixed pipeline
@@ -80,13 +83,21 @@ class DefaultContextPackBuilder(
// reorder freely and PromptRenderer can restore true turn order from the ordinal.
val stamped = entries.mapIndexed { index, entry -> entry.copy(ordinal = index) }
// A stalled stage re-issuing the exact same call otherwise burns budget on N identical
// (tool, arguments) pairs that all say the same thing — collapse those to the most recent
// occurrence. Fingerprint is (name, arguments) verbatim, so a call whose arguments genuinely
// differ (different start_line/end_line, different path) is a distinct read and untouched.
// Must run before FORMAT_COMPRESS below: that stage reencodes assistantToolCall content from
// raw ToolCallRequest JSON into a dotted-line format the fingerprint parser can't read.
val deduped = dedupeRepeatedToolCalls(stamped)
// Stage 1 (FORMAT_COMPRESS): lossless json→dotted-line compaction of structured entries.
val formatted = if (policy.enabled(CompressionStage.FORMAT_COMPRESS)) {
stamped.map { entry ->
deduped.map { entry ->
if (classifier.classify(entry) == ContextClass.STRUCTURED) reencode(entry, formatCompressor.compress(entry.content))
else entry
}
} else stamped
} else deduped
// Stage 3 (TOKEN_PRUNE): prune freeform prose, preserving protected spans. When TIER_SPLIT
// is on, the newest TIER0_TURNS freeform turns are left full-fidelity (tier 0).
@@ -151,7 +162,7 @@ class DefaultContextPackBuilder(
result
}
val retained = (pinned + compressed).sortedBy { it.ordinal }
val retained = reconcileToolPairs((pinned + compressed).sortedBy { it.ordinal })
val layers = retained.groupBy { it.layer }
val droppedCount = entries.size - retained.count { it.sourceType != "factSheet" }
val truncatedLayers = if (droppedCount > 0) {
@@ -235,6 +246,65 @@ class DefaultContextPackBuilder(
return rest + ranked
}
/**
* Collapses assistantToolCall/toolResult pairs that repeat an earlier call's exact (name,
* arguments) — keeping only the most recent occurrence — so a stalled stage repeating the same
* read doesn't show the model N copies of an identical question-and-answer. Content is the
* `ToolCallRequest` JSON built in SessionOrchestrator; parsed generically here (not via the
* `core:inference` type) to avoid a cross-core dependency. Calls whose arguments differ (e.g.
* different start_line/end_line, a different path) fingerprint differently and are untouched —
* that's genuine incremental progress, not repetition.
*/
private fun dedupeRepeatedToolCalls(entries: List<ContextEntry>): List<ContextEntry> {
val calls = entries.filter { it.sourceType == "assistantToolCall" }
val callsById = calls.associateBy { it.sourceId }
val fingerprints = calls.mapNotNull { entry ->
toolCallFingerprint(entry.content)?.let { entry.sourceId to it }
}
val dropIds = fingerprints.groupBy({ it.second }, { it.first })
.values
.filter { it.size > 1 }
.flatMapTo(mutableSetOf()) { sourceIds ->
val keepId = sourceIds.maxWith(compareBy(
{ id: String -> if (callsById.getValue(id).reasoning != null) 1 else 0 },
{ id: String -> callsById.getValue(id).ordinal }
))
sourceIds.filterNot { it == keepId }
}
if (dropIds.isEmpty()) return entries
return entries.filterNot {
(it.sourceType == "assistantToolCall" || it.sourceType == "toolResult") && it.sourceId in dropIds
}
}
private fun toolCallFingerprint(content: String): String? = runCatching {
val obj = Json.parseToJsonElement(content).jsonObject
val fn = obj["function"]?.jsonObject ?: return null
val name = fn["name"]?.jsonPrimitive?.content ?: return null
val arguments = fn["arguments"]?.jsonPrimitive?.content ?: ""
"$name:$arguments"
}.getOrNull()
/**
* A completed tool call is represented as two independently-IDed, independently-compressed
* entries — assistantToolCall + toolResult — linked only by [ContextEntry.sourceId]. Because
* `compressed` groups and trims by sourceType (see the `grouped` fold above), one side of a
* pair can survive its group's budget while the other is trimmed away. An orphaned
* assistantToolCall with no matching result renders as "I called this tool" with nothing after
* it — indistinguishable, to the model, from the call silently failing — which drove a live
* read-loop repro (2026-07-08: discovery stage re-issued the same file_read repeatedly once its
* result vanished from two consecutive rounds). Drop whichever side lost its pair so every
* tool-call entry that survives has its result, and vice versa.
*/
private fun reconcileToolPairs(entries: List<ContextEntry>): List<ContextEntry> {
val callIds = entries.filter { it.sourceType == "assistantToolCall" }.mapTo(mutableSetOf()) { it.sourceId }
val resultIds = entries.filter { it.sourceType == "toolResult" }.mapTo(mutableSetOf()) { it.sourceId }
return entries.filterNot {
(it.sourceType == "assistantToolCall" && it.sourceId !in resultIds) ||
(it.sourceType == "toolResult" && it.sourceId !in callIds)
}
}
private fun reencode(entry: ContextEntry, content: String): ContextEntry =
if (content == entry.content) entry else entry.copy(content = content, tokenEstimate = estimateTokens(content))
@@ -32,4 +32,10 @@ data class ContextEntry(
// Default OPTIONAL keeps prior behavior for callers that don't stamp buckets (router chat);
// the orchestrator stamps REQUIRED on task/constraint/input/failure entries.
val bucket: ContextBucket = ContextBucket.OPTIONAL,
// Chain-of-thought that preceded this ASSISTANT turn (assistantToolCall entries only), as
// captured on InferenceResponse.reasoning. Carried as a side channel, not JSON-encoded into
// `content`, so FORMAT_COMPRESS's dotted-line flatten of the tool-call JSON never touches it.
// Threaded back into the request as ChatMessage.reasoning so models that expect their own
// prior reasoning ahead of a tool-call turn (e.g. Gemma) don't degenerate into re-deriving it.
val reasoning: String? = null,
)
@@ -25,3 +25,18 @@ data class RepoKnowledgeRetrievedEvent(
val query: String,
val hits: List<RepoKnowledgeHit>,
) : EventPayload
/**
* Candidates that scored above the L3 index's own top-k cutoff but below the retriever's
* minimum-similarity floor, so were never rendered to the agent as "relevant." Without this,
* a hit below the floor is indistinguishable from one that was never a candidate — recorded
* per invariant #9 since the embedding-similarity score is itself an environment observation.
*/
@Serializable
@SerialName("RepoKnowledgeHitsFiltered")
data class RepoKnowledgeHitsFilteredEvent(
val sessionId: SessionId,
val query: String,
val threshold: Float,
val dropped: List<RepoKnowledgeHit>,
) : EventPayload
@@ -57,6 +57,7 @@ import com.correx.core.events.events.ModelUnloadedEvent
import com.correx.core.events.events.OrchestrationPausedEvent
import com.correx.core.events.events.OrchestrationResumedEvent
import com.correx.core.events.events.RefinementIterationEvent
import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent
import com.correx.core.events.events.RepoKnowledgeRetrievedEvent
import com.correx.core.events.events.RepoMapComputedEvent
import com.correx.core.events.events.RetryAttemptedEvent
@@ -154,6 +155,7 @@ val eventModule = SerializersModule {
subclass(RepoMapComputedEvent::class)
subclass(WorkspaceStateObservedEvent::class)
subclass(RepoKnowledgeRetrievedEvent::class)
subclass(RepoKnowledgeHitsFilteredEvent::class)
subclass(BriefGroundingCheckedEvent::class)
subclass(BriefEchoMismatchEvent::class)
subclass(StaticAnalysisCompletedEvent::class)
@@ -10,6 +10,14 @@ import kotlinx.serialization.Serializable
data class ChatMessage(
val role: String,
val content: String,
// Carries ContextEntry.reasoning forward for assistant tool-call turns — see that field's
// doc for why this exists (Gemma-family looping when its own prior reasoning is missing).
val reasoning: String? = null,
// Optional tool-call-id that links a role:tool response to the assistant's tool_calls entry
// (sourceId in the event store). Required by the Gemma4 Jinja template to render tool
// response blocks with the correct function name; without it every tool response is
// rendered as "unknown" and the thought/reasoning channel logic can break.
val toolCallId: String? = null,
)
object PromptRenderer {
@@ -61,5 +69,7 @@ object PromptRenderer {
EntryRole.USER -> "user"
},
content = content,
reasoning = reasoning,
toolCallId = if (role == EntryRole.TOOL) sourceId else null,
)
}
@@ -7,7 +7,7 @@ import com.correx.core.journal.model.DecisionRecord
class DecisionJournalRenderer(private val keepLast: Int = 40) {
fun render(state: DecisionJournalState, summaryText: String? = null): String {
if (state.records.isEmpty() && summaryText == null) return ""
val visible = foldResolvedRetries(state.records)
val visible = foldResolvedRetries(state.records).filter { it.kind in AGENT_RELEVANT_KINDS }
return buildString {
appendLine("## Decision history (chronological — ground truth)")
if (summaryText != null) {
@@ -16,7 +16,7 @@ class DecisionJournalRenderer(private val keepLast: Int = 40) {
appendLine("+${state.lowSalienceOmittedCount} routine transitions/retries omitted")
}
}
visible.takeLast(keepLast).forEach { appendLine("- [${it.kind}] ${it.summary}") }
visible.takeLast(keepLast).forEach { appendLine("- ${it.summary}") }
}.trimEnd()
}
@@ -46,4 +46,15 @@ class DecisionJournalRenderer(private val keepLast: Int = 40) {
record.copy(summary = "$stage: $count retr${if (count == 1) "y" else "ies"}, resolved — stage completed")
}
}
companion object {
/**
* What actually shapes an agent's next decision: the goal, human steering, and failures
* worth not repeating. APPROVAL/ARTIFACT/TRANSITION are bookkeeping for the audit trail —
* agents don't act on "Approval APPROVED (tier T2)" or "Advanced x → y", so they're
* dropped from the render (still present in [DecisionJournalState.records] for the journal).
*/
private val AGENT_RELEVANT_KINDS =
setOf(DecisionKind.INTENT, DecisionKind.STEERING, DecisionKind.PREEMPT, DecisionKind.FAILURE, DecisionKind.RETRY)
}
}
@@ -27,11 +27,28 @@ class DecisionJournalRendererTest {
records = listOf(record(1), record(2)),
)
val result = renderer.render(state)
assertTrue(result.contains("- [INTENT] summary-1"))
assertTrue(result.contains("- [INTENT] summary-2"))
assertTrue(result.contains("- summary-1"))
assertTrue(result.contains("- summary-2"))
assertFalse(result.contains("omitted"))
}
@Test
fun `approval, artifact and transition records are excluded from the render`() {
val state = DecisionJournalState(
records = listOf(
record(1, DecisionKind.APPROVAL, "Approval APPROVED (tier T2)"),
record(2, DecisionKind.ARTIFACT, "Validated artifact x at stage y"),
record(3, DecisionKind.TRANSITION, "Advanced a → b (t1)").copy(stageId = "b", fromStageId = "a"),
record(4, DecisionKind.FAILURE, "Stage b failed: boom"),
),
)
val result = renderer.render(state)
assertFalse(result.contains("Approval APPROVED"))
assertFalse(result.contains("Validated artifact"))
assertFalse(result.contains("Advanced a"))
assertTrue(result.contains("- Stage b failed: boom"))
}
@Test
fun `summaryText appears before records`() {
val state = DecisionJournalState(
@@ -42,7 +59,7 @@ class DecisionJournalRendererTest {
val lines = result.lines()
val headerIdx = lines.indexOfFirst { it.startsWith("## Decision history") }
val summaryIdx = lines.indexOfFirst { it == "Compaction summary here" }
val recordIdx = lines.indexOfFirst { it.startsWith("- [INTENT]") }
val recordIdx = lines.indexOfFirst { it.startsWith("- summary") }
assertTrue(headerIdx >= 0)
assertTrue(summaryIdx > headerIdx)
assertTrue(recordIdx > summaryIdx)
@@ -75,7 +92,7 @@ class DecisionJournalRendererTest {
assertTrue(result.contains("## Decision history"))
assertTrue(result.contains("Compacted summary"))
assertTrue(result.contains("+2 routine transitions/retries omitted"))
assertFalse(result.contains("- ["))
assertFalse(result.contains("- "))
}
@Test
@@ -90,8 +107,8 @@ class DecisionJournalRendererTest {
)
val result = renderer.render(state)
assertFalse(result.contains("did not produce"), "resolved retry detail must not render")
assertTrue(result.contains("- [RETRY] scaffold: 2 retries, resolved — stage completed"))
assertTrue(result.contains("- [TRANSITION] Advanced scaffold → hook"))
assertTrue(result.contains("- scaffold: 2 retries, resolved — stage completed"))
assertFalse(result.contains("Advanced scaffold"), "TRANSITION records are audit-only, not agent-relevant")
}
@Test
@@ -102,7 +119,7 @@ class DecisionJournalRendererTest {
),
)
val result = renderer.render(state)
assertTrue(result.contains("- [RETRY] Retry 1/3 of hook: did not produce"))
assertTrue(result.contains("- Retry 1/3 of hook: did not produce"))
}
@Test
@@ -115,6 +132,6 @@ class DecisionJournalRendererTest {
),
)
val result = renderer.render(state)
assertTrue(result.contains("- [RETRY] Retry 1/3 of scaffold: regression"))
assertTrue(result.contains("- Retry 1/3 of scaffold: regression"))
}
}
@@ -62,6 +62,7 @@ import com.correx.core.events.events.EventMetadata
import com.correx.core.events.events.EventPayload
import com.correx.core.events.events.InferenceCompletedEvent
import com.correx.core.events.events.RepoKnowledgeHit
import com.correx.core.events.events.InitialIntentEvent
import com.correx.core.events.events.RepoKnowledgeRetrievedEvent
import com.correx.core.events.events.RepoMapComputedEvent
import com.correx.core.events.events.InferenceFailedEvent
@@ -582,6 +583,11 @@ abstract class SessionOrchestrator(
var toolRounds = 0
var consecutiveReadOnlyRounds = 0
var consecutiveRejectedRounds = 0
// Fingerprints ("tool:arguments") of every read-only call made during the current
// no-write streak. A round only counts against the read-loop breaker if it repeats calls
// already seen — a stage reading N distinct files before writing is legitimate context
// gathering, not a loop, and shouldn't trip the same counter as re-reading the same file.
val seenReadFingerprints = mutableSetOf<String>()
// Set when the model produces its artifact via the emit_artifact tool instead of a final
// JSON message; overrides the post-loop capture of the (then-empty) assistant text.
var llmArtifactOverride: String? = null
@@ -697,7 +703,8 @@ abstract class SessionOrchestrator(
}
val toolEntries = dispatchToolCalls(sessionId, stageId,
inferenceResult.response.toolCalls, stageConfig, effectives,
approvalModeFor(session.state.boundProfile?.approvalMode))
approvalModeFor(session.state.boundProfile?.approvalMode),
inferenceResult.response.reasoning)
val fatalEntry = toolEntries.firstOrNull { it.content.startsWith("FATAL:") }
if (fatalEntry != null) {
emitProcessResultEvents(sessionId, stageId, stageConfig)
@@ -718,14 +725,23 @@ abstract class SessionOrchestrator(
if (owesFileWrite() &&
inferenceResult.response.toolCalls.none { it.function.name in WRITE_TOOL_NAMES }
) {
val roundFingerprints = inferenceResult.response.toolCalls
.map { "${it.function.name}:${it.function.arguments}" }
val madeProgress = roundFingerprints.any { it !in seenReadFingerprints }
seenReadFingerprints += roundFingerprints
if (madeProgress) {
consecutiveReadOnlyRounds = 0
} else {
consecutiveReadOnlyRounds++
if (consecutiveReadOnlyRounds >= READ_LOOP_NUDGE_THRESHOLD) {
consecutiveReadOnlyRounds = 0
inferenceResult = pushBack(readLoopNudge, forceWriteOnly = true)
continue
}
}
} else {
consecutiveReadOnlyRounds = 0
seenReadFingerprints.clear()
}
// Rejection-loop breaker (stage-type agnostic): every tool result this round was a
// rejection (plane-2 BLOCKED or a stage/policy ERROR) and nothing succeeded. A stage that
@@ -1028,10 +1044,14 @@ abstract class SessionOrchestrator(
// On a recoverable tool failure, append the tool's argument schema to the error fed back to the
// model so it can self-correct its next attempt (it already sees its own malformed call in the
// assistant entry above) rather than repeating the same mistake. Kept compact to bound context.
// NOTE: the "corrected arguments" phrasing was removed because it actively drives loops: a model
// that called list_dir frontend/ and got "directory not found" interprets "re-issue with corrected
// arguments" as "retry the same call" rather than "try a different path" — the arguments were
// valid, the path simply doesn't exist. The schema reference stays so malformed calls can still
// self-correct; the model must infer the tactic from the tool's error text.
private fun toolArgsHint(tool: Tool?): String =
tool?.let {
"\nThe '${it.name}' tool requires arguments matching this JSON schema — re-issue the " +
"call with corrected arguments: ${it.parametersSchema}"
"\nAccepted parameters for '${it.name}': ${it.parametersSchema}"
}.orEmpty()
private suspend fun dispatchToolCalls(
@@ -1041,11 +1061,18 @@ abstract class SessionOrchestrator(
stageConfig: StageConfig,
effectives: RunEffectives,
approvalMode: ApprovalMode,
// The reasoning that preceded this whole batch of tool calls (they all came from one
// inference response). Attached only to the first call's assistant entry — one thought
// block per model turn, matching how the response actually arrived.
reasoning: String? = null,
): List<ContextEntry> {
val executor = effectives.executor ?: return emptyList()
val processResultSlots = stageConfig.produces.filter { it.kind.id == "process_result" }
val fileWrittenSlots = stageConfig.produces.filter { it.kind.id == "file_written" }
return toolCalls.flatMap { toolCall ->
return toolCalls.withIndex().flatMap { (toolCallIndex, toolCall) ->
// Only the first call's assistant entry carries the reasoning — the whole batch came
// from one model turn/one thought block, so later calls in the same batch get none.
val toolCallReasoning = reasoning.takeIf { toolCallIndex == 0 }
val invocationId = ToolInvocationId(UUID.randomUUID().toString())
val parameters = parseToolArguments(toolCall.function.arguments)
val request = ToolRequest(
@@ -1099,6 +1126,7 @@ abstract class SessionOrchestrator(
content = Json.encodeToString(ToolCallRequest.serializer(), toolCall),
tokenEstimate = estimateTokens(toolCall.function.arguments),
role = EntryRole.ASSISTANT,
reasoning = toolCallReasoning,
),
ContextEntry(
id = ContextEntryId(UUID.randomUUID().toString()),
@@ -1155,6 +1183,7 @@ abstract class SessionOrchestrator(
content = Json.encodeToString(ToolCallRequest.serializer(), toolCall),
tokenEstimate = estimateTokens(toolCall.function.arguments),
role = EntryRole.ASSISTANT,
reasoning = toolCallReasoning,
),
ContextEntry(
id = ContextEntryId(UUID.randomUUID().toString()),
@@ -1223,6 +1252,7 @@ abstract class SessionOrchestrator(
content = Json.encodeToString(ToolCallRequest.serializer(), toolCall),
tokenEstimate = estimateTokens(toolCall.function.arguments),
role = EntryRole.ASSISTANT,
reasoning = toolCallReasoning,
),
ContextEntry(
id = ContextEntryId(UUID.randomUUID().toString()),
@@ -1282,6 +1312,7 @@ abstract class SessionOrchestrator(
content = Json.encodeToString(ToolCallRequest.serializer(), toolCall),
tokenEstimate = estimateTokens(toolCall.function.arguments),
role = EntryRole.ASSISTANT,
reasoning = toolCallReasoning,
),
ContextEntry(
id = ContextEntryId(UUID.randomUUID().toString()),
@@ -1353,6 +1384,7 @@ abstract class SessionOrchestrator(
content = Json.encodeToString(ToolCallRequest.serializer(), toolCall),
tokenEstimate = estimateTokens(toolCall.function.arguments),
role = EntryRole.ASSISTANT,
reasoning = toolCallReasoning,
)
val resultContent = when (result) {
is ToolResult.Success ->
@@ -1702,13 +1734,47 @@ abstract class SessionOrchestrator(
return closestPathByFilename(guessed, map.entries.map { it.path })
}
/**
* Resolves the actual prompt text a stage will see, for use as the repo-knowledge query.
* `metadata["prompt"]` holds a *path* to the prompt file (resolved into context separately,
* around line 431) — using that path string itself as the embedding query means the query the
* retriever compares against is a filesystem path, not the stage's real instructions, so
* similarity scores never reflect genuine relevance (2026-07-08 finding: no file scored a
* clear match, everything clustered in the embedder's baseline noise band for short strings).
*/
private fun resolveStagePromptText(stageId: StageId, stageConfig: StageConfig): String {
stageConfig.metadata["promptInline"]?.trim()?.takeIf { it.isNotBlank() }?.let { return it }
stageConfig.metadata["prompt"]?.let { path ->
runCatching { promptResolver.resolve(path) }.getOrNull()
?.trim()?.takeIf { it.isNotBlank() }
?.let { return it }
}
return stageId.value
}
/**
* The role's own prompt text (see [resolveStagePromptText]) is identical across every session
* that runs the stage — it carries zero session-specific signal, so similarity against it can
* only ever reflect generic lexical overlap with the role's boilerplate ("file_read", "ls",
* "grep"...), never what the user actually asked for. The one thing that IS session-specific
* is the initial user intent, so it's prepended here as the real anchor for the query (2026-07-08
* finding: without it, the query matched a tool script's own symbol names over the files the
* user's request actually named).
*/
private fun repoKnowledgeQuery(sessionId: SessionId, stageId: StageId, stageConfig: StageConfig): String {
val intent = eventStore.read(sessionId)
.mapNotNull { it.payload as? InitialIntentEvent }
.firstOrNull()?.intent?.trim()
val stagePrompt = resolveStagePromptText(stageId, stageConfig)
return if (intent.isNullOrBlank()) stagePrompt else "$intent\n\n$stagePrompt"
}
private suspend fun buildContextualRepoEntries(
sessionId: SessionId,
stageId: StageId,
stageConfig: StageConfig,
): List<ContextEntry> {
val stagePrompt = (stageConfig.metadata["promptInline"] ?: stageConfig.metadata["prompt"] ?: stageId.value)
.trim().ifBlank { stageId.value }
val stagePrompt = repoKnowledgeQuery(sessionId, stageId, stageConfig)
val recorded = eventStore.read(sessionId)
.mapNotNull { it.payload as? RepoKnowledgeRetrievedEvent }
.lastOrNull { it.stageId == stageId }
+21
View File
@@ -0,0 +1,21 @@
# Web UI Design Specification
## Aesthetic
- **Theme**: Soft Rounded + Blue Accent.
- **Consistency**: Should align with the TUI layout plan.
## Layout
- **Sidebar**: Centralized navigation.
- **Main Area**: Dynamic content based on selection.
- **Event Log**: Easily accessible (Source of Truth).
## Components
- **Cards**: For task summaries, status updates.
- **Badges**: For status (e.g., `CLAIMED`, `PENDING`, `COMPLETED`).
- **Progress Bars**: For multi-step task progress.
- **Service Layer**: Placeholders for `apps/server` endpoints.
## UX Priorities
- Clear hierarchy.
- Intuitive navigation.
- Avoid clutter (maintain information density).
@@ -85,6 +85,27 @@ Emit a JSON object that validates against the `execution_plan` schema:
Those fetch remote code and prompt on stdin; the run is unattended (no TTY) so they hang, and
`shell` rejects them outright. `shell` in a scaffold stage is for `npm install` / `npm run
build` on files that already exist, not for generating the project.
- **Every stage prompt must describe its exact JSON output structure when using a
structured kind.** The model that runs the stage never sees the raw JSON schema — it only sees
the kind's name (e.g. `analysis`, `research_report`, `review_report`) and the prompt you write
here. Therefore each stage prompt must:
- Conclude with a `Call \`emit_artifact\` with a JSON object matching this shape:` line
followed by the literal JSON structure the kind expects (fields, types, required vs
optional).
- When the kind is `file_written` no `emit_artifact` is needed — the stage writes files
directly. For all other kinds, the prompt must spell out the exact JSON schema inline.
- Include at least one concrete example JSON object matching the schema, so the model has
a template to follow.
- For reference, the available JSON schemas and their shapes are: `discovery`
(`{ready: boolean, questions: [{prompt, options?, multiSelect?, header?}]}`),
`analysis` (`{goal, constraints: [{description, severity}], risks:
[{description, severity, mitigation?}], open_questions: [{question, answered?}]}`),
`design` (`{approach, architecture: [{component, responsibility, interfaces}],
plan: [{step, action, files: [path]}]}`),
`impl_plan` (`{overview, steps: [{action, target, details}]}`),
`research_report` (`{summary, findings: [string], sources: [string]}`),
`review_report` (`{verdict, notes}`). Use these shapes verbatim in the
prompt — do not paraphrase or omit fields.
**edges** — describe every transition between stages. Rules:
- `from` and `to` must each be a declared stage `id` or the literal string `"done"`.
+18 -11
View File
@@ -1,23 +1,30 @@
You are the **Implementer**.
You receive the `impl_plan` artifact (above). Execute it using the tools available
(`file_read`, `file_write`, `file_edit`, and shell). File writes land in the bound workspace.
You receive the claimed task's acceptance criteria (injected above). Execute them using the tools
available (`file_read`, `file_write`, `file_edit`, and shell). File writes are auto-scoped to the
task's `affected_paths`; widen that scope via `propose_scope` when a needed path is outside it.
Steps:
1. If the analysis opened or referenced a task, `task_context` to load it and `task_update
action=claim` before you start (`task_create` one only if the work warrants tracking and none
exists). Skip this for a self-contained change.
2. Work through the plan `steps` in order. Read before you edit.
1. `task_context` to load the claimed task and confirm its acceptance criteria.
2. Work toward satisfying every acceptance criterion. Read before you edit.
3. Make the change with `file_write` / `file_edit`. Keep new code consistent with the
surrounding style, naming, and patterns.
4. Run the plan's `verification` commands (build/tests) via shell and fix what fails. Do not
4. **If a `file_write`/`file_edit` is BLOCKED** (the tool result starts with `BLOCKED:`), the
intended path is outside the claimed task's `affected_paths`. The kernel may append a
"Did you mean: <path>" suggestion — that is a string-similarity hint, **not** permission to
write there. Never work around a block by writing the file to a sibling path that happens to
be writable: that creates a shadow/duplicate file that breaks builds and confuses the
reviewer. Instead call `propose_scope` to add the intended path (or its enclosing glob) to
the manifest, then retry the write to the *original* path. If `propose_scope` is denied or
unavailable, stop and report the block in your output — do not invent scope.
5. Run the task's verification commands (build/tests) via shell and fix what fails. Do not
leave a step in a broken state.
5. When every step is done and verification passes, `task_update action=submit_for_review` on the
task (if any), then call the `stage_complete` tool.
6. When every criterion is met and verification passes, `task_update action=submit_for_review`
on the task, then call `stage_complete`.
The decision history above is ground truth. **If the reviewer requested changes** in a prior
round, you will see that verdict and its notes above — address those specific points; do not
re-do work that was already approved, and do not repeat a rejected approach.
Implement only what the plan calls for (YAGNI). If a step is genuinely blocked or the plan is
wrong, say so clearly rather than inventing scope.
Implement only what the task's acceptance criteria call for (YAGNI). If something is genuinely
blocked or the criteria are wrong, say so clearly rather than inventing scope.
+13 -11
View File
@@ -1,18 +1,21 @@
You are the **Reviewer** — the quality gate.
You receive the `patch` (the implementer's work), the `impl_plan` it was meant to satisfy, and
the `analysis` whose `requirements` are the **acceptance criteria** for this work.
You receive the `patch` (the implementer's work) and the `analysis` artifact (broader context).
First, use `task_context` to load the claimed task and read its `acceptance_criteria` — those
are the implementer's actual brief, not the full analysis.
Review the **diff against those acceptance criteria** — not whole files against personal taste.
The narrow question is: *does this change satisfy each stated requirement, correctly?* A finding
that doesn't map to a requirement, the plan, or a real defect is noise.
Review the **diff against those task acceptance criteria** — not whole files against personal
taste. The narrow question is: *does this change satisfy each stated criterion, correctly?* A
finding that doesn't map to a task criterion, an analysis requirement, or a real defect is noise.
The `analysis.requirements` are supplementary: use them for cross-checking, but the implementer
may be scoped to only a subset of the full analysis.
Review against the **current** ground truth, not a stale one: the decision history above
includes any user steering and your own prior verdicts. If the user steered the implementer
away from the original plan, judge against the steered direction — not the superseded plan.
Check:
1. Does the patch satisfy **every** requirement in `analysis.requirements` and step of `impl_plan`?
1. Does the patch satisfy **every** task acceptance criterion?
2. Is it correct, consistent with the codebase's patterns, and free of obvious defects?
3. Did verification (build/tests) actually pass?
4. No unrequested scope, no placeholders, no regressions.
@@ -21,16 +24,15 @@ Do **not** spend the review re-finding what deterministic tools already catch: c
formatter/detekt/lint violations, and failing tests are reported by the build itself. If
verification passed, trust it; if it failed, the failure is already on the record — point to it,
don't re-derive it. Your value is the judgment the tools can't give: whether the change actually
meets the requirements.
meets the acceptance criteria.
Emit your result as the `review_report` artifact (JSON, schema provided):
- `verdict`: exactly `"approved"` or `"changes_requested"`.
- `notes`: if `changes_requested`, list the specific, actionable changes needed (one per
line), each tied to the requirement or plan step it violates, so the implementer knows
exactly what to fix. If `approved`, briefly state which requirements the patch satisfies.
line), each tied to the task criterion or requirement it violates, so the implementer knows
exactly what to fix. If `approved`, briefly state which criteria the patch satisfies.
If the work is tracked as a task, reflect your verdict on it (use `task_context` first if you
need its own acceptance criteria): on `approved`, `task_update action=complete`; on
Reflect your verdict on the task: on `approved`, `task_update action=complete`; on
`changes_requested`, leave it claimed for the implementer — optionally add a `note` summarising
what's needed.
@@ -36,6 +36,12 @@ For `task_decompose`, provide:
`affected_paths` (workspace-relative globs), and `depends_on` (refs to other tasks in this batch
that must finish first).
**`affected_paths` globs must cover subdirectories recursively.** Use `**/*.tsx` instead of
`*.tsx`, `**/*.py` instead of `*.py`, etc. Sweep the actual directory tree with `list_dir` or
`shell find` before setting paths — if the target tree has subdirectories (e.g.
`components/ui/`, `components/layout/`), a non-recursive glob will jail the implementer to
only the top level and block writes to the real file locations.
Keep the graph small. Over-splitting is worse than under-splitting — a session works one task at a
time, and each child is a future claim. Aim for 25 children unless the request genuinely demands
more.
@@ -142,9 +142,15 @@ class LlamaCppInferenceProvider(
val baseMessages = PromptRenderer.render(request.contextPack)
val messages = baseMessages.map { msg ->
val toolCalls = if (msg.role == "assistant") {
parseAssistantToolCalls(msg.content)
} else emptyList()
LlamaCppChatMessage(
role = msg.role,
content = msg.content,
content = if (toolCalls.isNotEmpty()) null else msg.content,
reasoningContent = msg.reasoning,
toolCalls = toolCalls,
toolCallId = msg.toolCallId,
)
}
@@ -230,5 +236,40 @@ class LlamaCppInferenceProvider(
ProviderHealth.Unavailable("Health check failed: ${e.message}")
}
/**
* Parses an assistant message's content back into [ToolCallRequest] objects.
*
* The context builder's FORMAT_COMPRESS stage converts raw ToolCallRequest JSON into a compact
* dotted-line format (id = ..., function.name = ...). If compression is disabled, the content
* is still valid JSON. This handles both forms so the Gemma4 Jinja template sees the native
* `tool_calls` array — which it requires to render `reasoning_content` as a `<|channel>thought`
* block (without `tool_calls` the reasoning field is silently dropped and the model re-enters
* thought mode from scratch, causing the degenerate `<|channel>thought`~16384-token loop).
*/
private fun parseAssistantToolCalls(content: String): List<ToolCallRequest> {
// Try JSON parse first (FORMAT_COMPRESS disabled — raw ToolCallRequest JSON).
if (content.startsWith("{\"id\"")) {
runCatching {
return listOf(json.decodeFromString<ToolCallRequest>(content))
}
}
// Try dotted-line format (FORMAT_COMPRESS enabled).
if (!content.contains("function.name")) return emptyList()
val lines = content.split('\n')
var id: String? = null
var name: String? = null
var arguments: String? = null
for (line in lines) {
when {
line.startsWith("id = ") -> id = line.removePrefix("id = ").trim()
line.startsWith("function.name = ") -> name = line.removePrefix("function.name = ").trim()
line.startsWith("function.arguments = ") -> arguments = line.removePrefix("function.arguments = ").trim()
}
}
return if (name != null && arguments != null) {
listOf(ToolCallRequest(id = id, function = ToolCallFunction(name = name, arguments = arguments)))
} else emptyList()
}
override fun capabilities(): Set<CapabilityScore> = descriptor.capabilities
}
@@ -88,7 +88,7 @@ class OpenAiCompatInferenceProvider(
// tool_calls turn (we don't carry tool_call_ids through the context pack). Fold any
// non-standard role into a user turn so the model still sees the content.
if (msg.role in STANDARD_ROLES) {
OpenAiChatMessage(role = msg.role, content = msg.content)
OpenAiChatMessage(role = msg.role, content = msg.content, reasoningContent = msg.reasoning)
} else {
OpenAiChatMessage(role = "user", content = "[${msg.role}] ${msg.content}")
}
@@ -135,8 +135,18 @@ class FileEditTool(
operation == "patch" && !request.parameters.containsKey("patch") ->
ValidationResult.Invalid("Missing 'patch' parameter for 'patch' operation.")
// Don't point the model at 'file_read'/'list_dir' here — this rejection is commonly hit
// in exactly the round those tools have been deliberately withheld (the read-loop
// breaker forces a write-only round after repeated reads with no write), so telling it
// to go use tools that aren't in its tool list this round would just compound the stall.
operation == "read" ->
ValidationResult.Invalid(
"'edit_file' has no 'read' operation. It only supports 'append', 'replace', or 'patch' — " +
"proceed with one of those using what you already know about the file.",
)
operation !in listOf("append", "replace", "patch") ->
ValidationResult.Invalid("Unknown operation: $operation")
ValidationResult.Invalid("Unknown operation: $operation. Expected 'append', 'replace', or 'patch'.")
else -> ValidationResult.Valid
}
@@ -154,6 +154,7 @@ class FileWriteTool(
val originalContent = runCatching { Files.readString(path) }.getOrDefault("")
val result = runCatching {
Files.createDirectories(path.parent)
AtomicFileWriter.write(path, content.toByteArray(Charsets.UTF_8))
ToolResult.Success(
invocationId = request.invocationId,
@@ -97,8 +97,17 @@ class ListDirTool(
val recursive = request.parameters["recursive"]?.toString()?.toBoolean() ?: false
val root = resolvePath(pathString)
when {
!Files.exists(root) ->
ToolResult.Failure(request.invocationId, "Directory not found: $pathString", recoverable = true)
!Files.exists(root) -> {
val suggestions = similarEntries(root)
val msg = buildString {
append("Directory not found: $pathString")
if (suggestions.isNotEmpty()) {
append(". Did you mean ")
append(suggestions.joinToString(", ", postfix = "?"))
}
}
ToolResult.Failure(request.invocationId, msg, recoverable = true)
}
!Files.isDirectory(root) ->
ToolResult.Failure(request.invocationId, "Not a directory: $pathString", recoverable = true)
else -> runCatching { walk(root, recursive, request) }.getOrElse {
@@ -109,8 +118,13 @@ class ListDirTool(
private fun walk(root: Path, recursive: Boolean, request: ToolRequest): ToolResult {
val matcher = GitignoreMatcher(workingDir ?: root, root)
// The caller explicitly targeted `root` — if a rule already covers it (e.g. an entire
// subtree gitignored as scaffold output), don't let that same rule keep pruning everything
// found underneath; other, unrelated rules (node_modules/ inside it, etc.) still apply.
matcher.exemptRulesMatching(root)
val out = ArrayList<String>()
var truncated = false
var ignoredCount = 0
// Emit one entry; TERMINATE the walk once the cap is hit so a huge tree can't run away.
fun emit(path: Path, isDir: Boolean): FileVisitResult {
val slash = if (isDir) "/" else ""
@@ -121,8 +135,11 @@ class ListDirTool(
val maxDepth = if (recursive) Int.MAX_VALUE else 1
Files.walkFileTree(root, emptySet(), maxDepth, object : SimpleFileVisitor<Path>() {
override fun preVisitDirectory(dir: Path, attrs: BasicFileAttributes): FileVisitResult {
if (dir != root && (dir.name == ".git" || matcher.ignored(dir, isDir = true)))
if (dir != root && dir.name == ".git") return FileVisitResult.SKIP_SUBTREE
if (dir != root && matcher.ignored(dir, isDir = true)) {
ignoredCount++
return FileVisitResult.SKIP_SUBTREE
}
matcher.loadFrom(dir)
return if (dir == root) FileVisitResult.CONTINUE else emit(dir, isDir = true)
}
@@ -130,15 +147,83 @@ class ListDirTool(
// A directory sitting exactly at maxDepth is delivered here (walkFileTree won't
// descend it), so honour its dir-ness for both the ignore check and the trailing '/'.
val isDir = attrs.isDirectory
return if (matcher.ignored(file, isDir)) FileVisitResult.CONTINUE else emit(file, isDir)
if (matcher.ignored(file, isDir)) {
ignoredCount++
return FileVisitResult.CONTINUE
}
return emit(file, isDir)
}
override fun visitFileFailed(file: Path, exc: java.io.IOException) = FileVisitResult.CONTINUE
})
out.sort()
val body = out.joinToString("\n").ifEmpty { "(empty)" }
val note =
if (truncated) "\n… truncated at $MAX_ENTRIES entries; narrow the path or list a sub-directory." else ""
return ToolResult.Success(request.invocationId, output = body + note)
return ToolResult.Success(request.invocationId, output = render(root, out, truncated, ignoredCount))
}
/**
* `<path>/<type>/<entries>` wrapper (matching the shape other coding-agent tools use) instead of
* a bare newline list — the count and explicit `directory` tag let a small model tell "empty
* listing" from "malformed output" and know how many entries to expect. Notes on truncation/
* gitignore-pruning are surfaced here too, so a shorter-than-expected listing isn't mistaken for
* the real tree.
*/
private fun render(root: Path, entries: List<String>, truncated: Boolean, ignoredCount: Int): String {
val notes = buildList {
if (truncated) add("truncated at $MAX_ENTRIES entries; narrow the path or list a sub-directory")
if (ignoredCount > 0) add("$ignoredCount ${if (ignoredCount == 1) "entry" else "entries"} hidden by .gitignore")
}
return buildString {
appendLine("<path>$root</path>")
appendLine("<type>directory</type>")
appendLine("<entries>")
appendLine(entries.joinToString("\n").ifEmpty { "(empty)" })
if (notes.isNotEmpty()) appendLine(notes.joinToString("; "))
append("(${entries.size} ${if (entries.size == 1) "entry" else "entries"})")
appendLine()
append("</entries>")
}
}
private fun similarEntries(target: Path): List<String> {
val parent = target.parent ?: return emptyList()
if (!Files.isDirectory(parent)) return emptyList()
val targetName = target.fileName?.toString() ?: return emptyList()
return try {
Files.list(parent).use { stream ->
stream.map { it.fileName.toString() }
.filter { it != targetName && isSimilarName(it, targetName) }
.toList()
.sortedWith(compareByDescending<String> { commonPrefixLen(it, targetName) }.thenBy { it })
.take(3)
}
} catch (_: Exception) { emptyList() }
}
private fun isSimilarName(a: String, b: String): Boolean {
val prefix = commonPrefixLen(a, b)
return prefix >= 2 || levenshtein(a, b) <= 3
}
private fun commonPrefixLen(a: String, b: String): Int {
var i = 0
val minLen = minOf(a.length, b.length)
while (i < minLen && a[i].lowercaseChar() == b[i].lowercaseChar()) i++
return i
}
private fun levenshtein(a: String, b: String): Int {
val dp = Array(a.length + 1) { IntArray(b.length + 1) }
for (i in 0..a.length) dp[i][0] = i
for (j in 0..b.length) dp[0][j] = j
for (i in 1..a.length) {
for (j in 1..b.length) {
dp[i][j] = minOf(
dp[i - 1][j] + 1,
dp[i][j - 1] + 1,
dp[i - 1][j - 1] + if (a[i - 1].lowercaseChar() == b[i - 1].lowercaseChar()) 0 else 1
)
}
}
return dp[a.length][b.length]
}
private companion object {
@@ -176,10 +261,17 @@ internal class GitignoreMatcher(anchor: Path, target: Path) {
}
}
fun ignored(path: Path, isDir: Boolean): Boolean = rules.any { r ->
fun ignored(path: Path, isDir: Boolean): Boolean = rules.any { matches(it, path, isDir) }
/** Drop whichever rules currently cover [path] — used so an explicitly-targeted (already
* gitignored) listing root doesn't have that same rule keep pruning its own children. */
fun exemptRulesMatching(path: Path) {
rules.removeAll { matches(it, path, isDir = true) }
}
private fun matches(r: Rule, path: Path, isDir: Boolean): Boolean =
(!r.dirOnly || isDir) && path.startsWith(r.base) &&
r.regex.matches(r.base.relativize(path).toString().replace('\\', '/'))
}
private fun compile(base: Path, raw: String): Rule {
var pat = raw
@@ -41,7 +41,7 @@ object PlanLinter {
fun lint(graph: WorkflowGraph, seeds: Set<String> = seedArtifacts): PlanLintResult =
PlanLintResult(
hardFailures = unproducedNeeds(graph, seeds) + trapStates(graph),
hardFailures = unproducedNeeds(graph, seeds) + trapStates(graph) + unreferencedPromptArtifacts(graph, seeds),
softFindings = stageCount(graph) + fanOut(graph) + emptyBriefs(graph) + duplicateBriefs(graph),
)
@@ -89,6 +89,35 @@ object PlanLinter {
}
}
/**
* H3: a stage's inline prompt mentions an artifact ID that is produced by the plan or
* provided by a seed, but the stage does not declare it in its `needs`. The stage will
* not have access to that artifact at runtime, so any reference in the prompt is a dangling
* assumption that will confuse or stall the model — force the architect to either add the
* artifact to `needs` or reword the prompt.
*/
private fun unreferencedPromptArtifacts(graph: WorkflowGraph, seeds: Set<String>): List<PlanLintFinding> {
val produced = graph.stages.values.flatMap { it.produces }.map { it.name.value }.toSet() + seeds
return graph.stages.entries.flatMap { (id, stage) ->
val prompt = briefOf(stage)
val needIds = stage.needs.map { it.value }.toSet()
val ownProduces = stage.produces.map { it.name.value }.toSet()
produced.filter { prodId ->
prompt.isNotBlank() &&
prodId.length >= 2 && // skip single-char IDs (false-positive risk with substring matching)
Regex("\\b${Regex.escape(prodId)}\\b").containsMatchIn(prompt) &&
prodId !in needIds &&
prodId !in ownProduces
}.map { prodId ->
PlanLintFinding(
code = "unreferenced_prompt_artifact",
stageId = id.value,
detail = "prompt mentions '$prodId' but it is not declared in needs",
)
}
}
}
/** S1: more stages than the ceiling. */
private fun stageCount(graph: WorkflowGraph): List<PlanLintFinding> =
if (graph.stages.size > STAGE_CEILING) {
@@ -92,6 +92,37 @@ class PlanLinterTest {
assertTrue(result.passed, "analysis is a planning-phase seed, not an unproduced need")
}
@Test
fun `a stage mentioning its own produced artifact in the prompt is not a failure`() {
val g = graph(
stages = mapOf(
"a" to stage(produces = listOf("plan"), prompt = "write the plan artifact"),
"b" to stage(needs = listOf("plan"), prompt = "build"),
),
edges = listOf("a" to "b", "b" to "done"),
start = "a",
)
val result = PlanLinter.lint(g)
assertTrue(result.passed, "expected pass, hard=${result.hardFailures}")
}
@Test
fun `a prompt mentioning another stage's artifact not in needs is a hard failure`() {
val g = graph(
stages = mapOf(
"a" to stage(produces = listOf("plan"), prompt = "write the plan artifact"),
"b" to stage(produces = listOf("report"), prompt = "use plan to build"),
),
edges = listOf("a" to "b", "b" to "done"),
start = "a",
)
val result = PlanLinter.lint(g)
assertFalse(result.passed)
val h = result.hardFailures.single { it.code == "unreferenced_prompt_artifact" }
assertEquals("b", h.stageId)
assertTrue(h.detail.contains("plan"))
}
@Test
fun `a cycle with no exit is a trap-state hard failure`() {
val g = graph(
@@ -134,10 +134,10 @@ class DefaultContextPackBuilderTest {
// layer reads call, result, call, result — not all calls then all results.
val entries = listOf(
entry("task", ContextLayer.L1, 10).copy(sourceType = "agentPrompt"),
entry("call1", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall"),
entry("result1", ContextLayer.L2, 10).copy(sourceType = "toolResult"),
entry("call2", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall"),
entry("result2", ContextLayer.L2, 10).copy(sourceType = "toolResult"),
entry("call1", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall", sourceId = "tool1"),
entry("result1", ContextLayer.L2, 10).copy(sourceType = "toolResult", sourceId = "tool1"),
entry("call2", ContextLayer.L2, 10).copy(sourceType = "assistantToolCall", sourceId = "tool2"),
entry("result2", ContextLayer.L2, 10).copy(sourceType = "toolResult", sourceId = "tool2"),
)
val pack = builder.build(packId, sessionId, stageId, entries, TokenBudget(limit = 4000))
val l2Ids = pack.layers[ContextLayer.L2]!!.map { it.id.value }
@@ -18,12 +18,13 @@ class PromptRendererOrderingTest {
private val sessionId = SessionId("s1")
private val stageId = StageId("stage-1")
private fun entry(id: String, layer: ContextLayer, role: EntryRole, sourceType: String) = ContextEntry(
private fun entry(id: String, layer: ContextLayer, role: EntryRole, sourceType: String, sourceId: String = id) =
ContextEntry(
id = ContextEntryId(id),
layer = layer,
content = id,
sourceType = sourceType,
sourceId = id,
sourceId = sourceId,
tokenEstimate = 10,
role = role,
)
@@ -36,10 +37,10 @@ class PromptRendererOrderingTest {
val entries = listOf(
entry("sys", ContextLayer.L0, EntryRole.SYSTEM, "systemPrompt"),
entry("task", ContextLayer.L1, EntryRole.USER, "agentPrompt"),
entry("call1", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall"),
entry("result1", ContextLayer.L2, EntryRole.TOOL, "toolResult"),
entry("call2", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall"),
entry("result2", ContextLayer.L2, EntryRole.TOOL, "toolResult"),
entry("call1", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall", sourceId = "tool1"),
entry("result1", ContextLayer.L2, EntryRole.TOOL, "toolResult", sourceId = "tool1"),
entry("call2", ContextLayer.L2, EntryRole.ASSISTANT, "assistantToolCall", sourceId = "tool2"),
entry("result2", ContextLayer.L2, EntryRole.TOOL, "toolResult", sourceId = "tool2"),
)
val pack = builder.build(ContextPackId("p"), sessionId, stageId, entries, TokenBudget(limit = 4000))