diff --git a/apps/cli/src/main/kotlin/com/correx/apps/cli/commands/StatsCommand.kt b/apps/cli/src/main/kotlin/com/correx/apps/cli/commands/StatsCommand.kt index adc12fa4..134d0f1d 100644 --- a/apps/cli/src/main/kotlin/com/correx/apps/cli/commands/StatsCommand.kt +++ b/apps/cli/src/main/kotlin/com/correx/apps/cli/commands/StatsCommand.kt @@ -54,6 +54,18 @@ data class TierApprovalStatsDto( val avgWaitMs: Long, ) +@Serializable +data class QualityStatsDto( + val sourceFetches: Long = 0, + val lowQualityExtractions: Long = 0, + val extractionQualityRate: Double = 1.0, + val retrievedHits: Long = 0, + val droppedHits: Long = 0, + val retrievalPrecision: Double = 1.0, + val briefDriftCount: Long = 0, + val capabilityGapCount: Long = 0, +) + @Serializable data class StatsReportDto( val sessionId: String, @@ -75,6 +87,7 @@ data class StatsReportDto( val avgApprovalWaitMs: Long = 0, val perTier: List = emptyList(), val failures: FailureMetricsDto = FailureMetricsDto(), + val quality: QualityStatsDto = QualityStatsDto(), val inferencePct: Double = 0.0, val toolPct: Double = 0.0, val approvalWaitPct: Double = 0.0, @@ -146,6 +159,17 @@ fun renderStats(report: StatsReportDto): String { "stage: ${f.stageFailures} workflow: ${f.workflowFailures}" lines += "" + val q = report.quality + lines += "Signal quality" + lines += " extraction: %d fetched, %d low-quality (%.0f%% clean)".format( + Locale.ROOT, q.sourceFetches, q.lowQualityExtractions, q.extractionQualityRate * 100.0, + ) + lines += " retrieval: %d kept, %d filtered (%.0f%% precision)".format( + Locale.ROOT, q.retrievedHits, q.droppedHits, q.retrievalPrecision * 100.0, + ) + lines += " brief drift: ${q.briefDriftCount} capability gaps: ${q.capabilityGapCount}" + lines += "" + lines += "Time accounting (% of session wall time)" lines += " inference: %.1f%% tools: %.1f%% approval-wait: %.1f%% other: %.1f%%".format( Locale.ROOT, diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt index ab0099d5..51aa1ff0 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/Main.kt @@ -82,6 +82,7 @@ import com.correx.core.events.types.SessionId import com.correx.infrastructure.InfrastructureModule import com.correx.infrastructure.inference.DefaultProviderRegistry import com.correx.infrastructure.workflow.ExecutionPlanCompiler +import com.correx.infrastructure.workflow.PlanLinter import com.correx.infrastructure.inference.CapabilityAwareRoutingStrategy import com.correx.infrastructure.inference.commons.ManagedInferenceRouter import com.correx.infrastructure.inference.commons.AmdResourceProbe @@ -388,7 +389,17 @@ fun main() { repositories = repositories, engines = engines.copy( planCompilationCheck = { json, candidateId -> - runCatching { planCompiler.compile(json, candidateId) }.exceptionOrNull()?.message + // Compile first; a compile failure returns its message. If it compiles, also run the + // deterministic lint here so a hard lint failure (unproduced need, trap state) feeds + // back to the architect via the same retry loop — rather than the post-planning + // FreestyleDriver lint rejecting it after the session already reached COMPLETED. + val compiled = runCatching { planCompiler.compile(json, candidateId) } + compiled.exceptionOrNull()?.message + ?: PlanLinter.lint(compiled.getOrThrow()).takeIf { !it.passed }?.let { lint -> + "plan lint failed: " + lint.hardFailures.joinToString("; ") { + "${it.code}${it.stageId?.let { s -> " @$s" } ?: ""}: ${it.detail}" + } + } }, ), retryCoordinator = DefaultRetryCoordinator(eventStore), diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt index 14447809..ee6fafa8 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/DomainEventMapper.kt @@ -18,6 +18,7 @@ import com.correx.core.events.events.ChatSessionStartedEvent import com.correx.core.events.events.ChatTurnEvent import com.correx.core.events.events.ClarificationRequestedEvent import com.correx.core.events.events.InferenceCompletedEvent +import com.correx.core.events.events.SessionNamedEvent import com.correx.core.events.events.SessionWorkspaceBoundEvent import com.correx.core.events.events.WorkflowStartedEvent import com.correx.core.events.events.InferenceFailedEvent @@ -90,6 +91,13 @@ suspend fun domainEventToServerMessage( sessionSequence = sessionSequence, ) + is SessionNamedEvent -> ServerMessage.SessionRenamed( + sessionId = p.sessionId, + name = p.name, + sequence = seq, + sessionSequence = sessionSequence, + ) + is ChatTurnEvent -> ServerMessage.ChatTurn( sessionId = p.sessionId, turnId = p.turnId, @@ -182,6 +190,7 @@ suspend fun domainEventToServerMessage( sessionId = p.sessionId, toolName = p.toolName, tier = p.tier, + params = prettyToolParams(p.request.parameters), sequence = seq, sessionSequence = sessionSequence, ) @@ -354,14 +363,19 @@ private suspend fun mapInferenceCompleted( p: InferenceCompletedEvent, artifactStore: ArtifactStore, sessionSequence: Long, -): ServerMessage = runCatching { - artifactStore.get(p.responseArtifactId)?.toString(Charsets.UTF_8) ?: "" -}.getOrElse { "" }.run { - ServerMessage.InferenceCompleted( +): ServerMessage { + val response = runCatching { + artifactStore.get(p.responseArtifactId)?.toString(Charsets.UTF_8) ?: "" + }.getOrElse { "" } + val reasoning = p.reasoningArtifactId?.let { id -> + runCatching { artifactStore.get(id)?.toString(Charsets.UTF_8) ?: "" }.getOrElse { "" } + } ?: "" + return ServerMessage.InferenceCompleted( sessionId = p.sessionId, stageId = p.stageId, - outputSummary = this, - responseText = this, + outputSummary = response, + responseText = response, + reasoning = reasoning, occurredAt = event.metadata.timestamp.toEpochMilliseconds(), totalTokens = p.tokensUsed.totalTokens, sequence = event.sequence, @@ -389,3 +403,33 @@ private fun mapApprovalRequested( sequence = seq, sessionSequence = sessionSequence, ) + +/** + * Render a tool call's raw parameter map into compact, human-readable "key=value" cells for the + * client's tool-call row. Primary keys (path/command/query/…) lead so the most salient argument is + * first; long values are truncated and multi-line values collapsed so the WS frame stays small. + */ +internal fun prettyToolParams(parameters: Map): List { + if (parameters.isEmpty()) return emptyList() + val primary = listOf("command", "path", "query", "url", "pattern", "content", "operation") + val ordered = parameters.entries.sortedWith( + compareBy({ primary.indexOf(it.key).let { i -> if (i < 0) primary.size else i } }, { it.key }), + ) + return ordered.take(MAX_PARAM_CELLS).map { (k, v) -> "$k=${formatParamValue(v)}" } +} + +private fun formatParamValue(value: Any?): String { + val raw = when (value) { + null -> "null" + is String -> value + is Collection<*> -> value.joinToString(", ", prefix = "[", postfix = "]") { formatParamValue(it) } + else -> value.toString() + } + val flattened = raw.replace('\n', ' ').replace('\r', ' ').trim() + val clipped = if (flattened.length > MAX_PARAM_VALUE_LEN) flattened.take(MAX_PARAM_VALUE_LEN) + "…" else flattened + // Quote strings that carry whitespace so the boundary of the value is unambiguous in the row. + return if (value is String && clipped.any { it.isWhitespace() }) "\"$clipped\"" else clipped +} + +private const val MAX_PARAM_CELLS = 5 +private const val MAX_PARAM_VALUE_LEN = 80 diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/SessionEventBridge.kt b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/SessionEventBridge.kt index 925b4bd3..0eee9e72 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/bridge/SessionEventBridge.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/bridge/SessionEventBridge.kt @@ -8,6 +8,7 @@ import com.correx.apps.server.protocol.SessionStateDto import com.correx.apps.server.protocol.StageToolDecl import com.correx.apps.server.protocol.ToolDecl import com.correx.apps.server.protocol.ToolRecordDto +import com.correx.apps.server.protocol.TranscriptRowDto import com.correx.apps.server.registry.WorkflowRegistry import com.correx.core.approvals.ApprovalProjector import com.correx.core.approvals.DefaultApprovalReducer @@ -20,6 +21,7 @@ import com.correx.core.events.events.InferenceTimeoutEvent import com.correx.core.events.events.NewEvent import com.correx.core.events.events.ApprovalRequestedEvent import com.correx.core.events.events.ArtifactCreatedEvent +import com.correx.core.events.events.ChatTurnEvent import com.correx.core.events.events.ClarificationAnsweredEvent import com.correx.core.events.events.ClarificationRequestedEvent import com.correx.core.events.events.OrchestrationPausedEvent @@ -28,6 +30,7 @@ import com.correx.core.events.events.StageCompletedEvent import com.correx.core.events.events.SessionWorkspaceBoundEvent import com.correx.core.events.events.StageFailedEvent import com.correx.core.events.events.StoredEvent +import com.correx.core.events.events.TalkieNarrationEvent import com.correx.core.events.events.ToolExecutionCompletedEvent import com.correx.core.events.events.ToolExecutionFailedEvent import com.correx.core.events.events.ToolExecutionRejectedEvent @@ -166,6 +169,7 @@ class SessionEventBridge( } val toolRecords = rebuildTools(events) + val transcript = reconstructTranscript(events) send(ServerMessage.SessionSnapshot( sessionId = sessionId, @@ -178,6 +182,7 @@ class SessionEventBridge( pendingApprovals = pendingApprovals, tools = toolRecords, recentEvents = recentEvents, + transcript = transcript, lastOutput = lastOutput, lastResponse = lastOutput, workspaceRoot = workspaceRoot, @@ -233,6 +238,67 @@ class SessionEventBridge( }?.let { stored -> mapper.map(stored, stored.sessionSequence)?.let { send(it) } } } + /** + * Rebuild the OUTPUT-transcript source rows (chat / narration / thinking / tool) so a reopened + * session restores its full conversation. Ships raw content only — the client applies its own + * diff/param/action formatting. Reasoning is resolved from CAS (same as the live path). Tool + * params are paired to their completion by invocation id. + */ + private suspend fun reconstructTranscript(events: List): List { + val paramsByInvocation = mutableMapOf>() + val rows = mutableListOf() + for (event in events) { + when (val p = event.payload) { + is ChatTurnEvent -> rows.add( + TranscriptRowDto( + kind = "chat", + role = p.role.name, + content = p.content, + latencyMs = p.latencyMs, + totalTokens = p.tokensUsed?.totalTokens, + ), + ) + is TalkieNarrationEvent -> rows.add( + TranscriptRowDto( + kind = "narration", + content = p.content, + latencyMs = p.latencyMs, + totalTokens = p.tokensUsed?.totalTokens, + ), + ) + is InferenceCompletedEvent -> { + val reasoning = p.reasoningArtifactId + ?.let { runCatching { artifactStore.get(it)?.toString(Charsets.UTF_8) }.getOrNull() } + .orEmpty() + if (reasoning.isNotBlank()) { + rows.add(TranscriptRowDto(kind = "thinking", content = reasoning)) + } + } + is ToolInvocationRequestedEvent -> + paramsByInvocation[p.invocationId.value] = prettyToolParams(p.request.parameters) + is ToolExecutionCompletedEvent -> rows.add( + TranscriptRowDto( + kind = "tool", + status = "completed", + toolName = p.toolName, + diff = p.receipt.diff, + params = paramsByInvocation[p.invocationId.value].orEmpty(), + summary = p.receipt.outputSummary, + affectedEntities = p.receipt.affectedEntities, + ), + ) + is ToolExecutionFailedEvent -> rows.add( + TranscriptRowDto(kind = "tool", status = "failed", toolName = p.toolName, summary = p.reason), + ) + is ToolExecutionRejectedEvent -> rows.add( + TranscriptRowDto(kind = "tool", status = "rejected", toolName = p.toolName, summary = p.reason), + ) + else -> { /* not a transcript row */ } + } + } + return rows + } + private fun eventToEntry(event: StoredEvent): EventEntryDto? { val ts = event.metadata.timestamp.toEpochMilliseconds() return when (val p = event.payload) { diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsInspectionService.kt b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsInspectionService.kt index 8f5c9aad..3e315d9a 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsInspectionService.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsInspectionService.kt @@ -34,6 +34,24 @@ data class TierApprovalStatsDto( val avgWaitMs: Long, ) +/** + * Signal-quality view: raw counts plus the two derived ratios an operator reads. Extraction + * quality is the share of research fetches that cleared the quality bar; retrieval precision is + * the share of repo-knowledge hits kept (not filtered below the similarity threshold). Both are + * 1.0 when there's nothing to judge (no fetches / no retrieval), so a clean run reads as clean. + */ +@Serializable +data class QualityStatsDto( + val sourceFetches: Long = 0, + val lowQualityExtractions: Long = 0, + val extractionQualityRate: Double = 1.0, + val retrievedHits: Long = 0, + val droppedHits: Long = 0, + val retrievalPrecision: Double = 1.0, + val briefDriftCount: Long = 0, + val capabilityGapCount: Long = 0, +) + /** * Render-friendly view of a session's metrics. Raw sums plus the derived ratios the operator * actually reads; the time-accounting percentages split session wall time into inference / @@ -61,6 +79,7 @@ data class MetricsReport( val avgApprovalWaitMs: Long, val perTier: List, val failures: FailureMetrics, + val quality: QualityStatsDto = QualityStatsDto(), val inferencePct: Double, val toolPct: Double, val approvalWaitPct: Double, @@ -125,6 +144,7 @@ class MetricsInspectionService(private val eventStore: EventStore) { ) }, failures = s.failures, + quality = qualityStats(s.quality), inferencePct = pct(s.inference.totalLatencyMs, durationMs), toolPct = pct(s.tools.totalDurationMs, durationMs), approvalWaitPct = pct(s.approvals.totalWaitMs, durationMs), @@ -145,4 +165,20 @@ class MetricsInspectionService(private val eventStore: EventStore) { private fun avg(total: Long, count: Long): Long = if (count <= 0L) 0L else total / count + + private fun qualityStats(q: QualityMetrics): QualityStatsDto { + val keptExtractions = (q.sourceFetches - q.lowQualityExtractions).coerceAtLeast(0L) + val totalHits = q.retrievedHits + q.droppedHits + return QualityStatsDto( + sourceFetches = q.sourceFetches, + lowQualityExtractions = q.lowQualityExtractions, + // 1.0 when nothing was fetched — no evidence of low quality reads as clean. + extractionQualityRate = if (q.sourceFetches <= 0L) 1.0 else keptExtractions.toDouble() / q.sourceFetches, + retrievedHits = q.retrievedHits, + droppedHits = q.droppedHits, + retrievalPrecision = if (totalHits <= 0L) 1.0 else q.retrievedHits.toDouble() / totalHits, + briefDriftCount = q.briefDriftCount, + capabilityGapCount = q.capabilityGapCount, + ) + } } diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsProjection.kt b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsProjection.kt index 116a9fce..ff6813dc 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsProjection.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsProjection.kt @@ -2,9 +2,17 @@ package com.correx.apps.server.metrics import com.correx.core.events.events.ApprovalDecisionResolvedEvent import com.correx.core.events.events.ApprovalRequestedEvent +import com.correx.core.events.events.BriefEchoMismatchEvent +import com.correx.core.events.events.CapabilityGapReflectedEvent +import com.correx.core.events.events.CapabilityGapVerdict +import com.correx.core.events.events.EventPayload import com.correx.core.events.events.InferenceCompletedEvent import com.correx.core.events.events.InferenceFailedEvent import com.correx.core.events.events.InferenceTimeoutEvent +import com.correx.core.events.events.LowQualityExtractionEvent +import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent +import com.correx.core.events.events.RepoKnowledgeRetrievedEvent +import com.correx.core.events.events.SourceFetchedEvent import com.correx.core.events.events.StageFailedEvent import com.correx.core.events.events.StoredEvent import com.correx.core.events.events.ToolExecutionCompletedEvent @@ -44,13 +52,37 @@ class MetricsProjection : Projection { is ApprovalDecisionResolvedEvent -> spanned.applyApprovalResolved(payload) is StageFailedEvent -> spanned.bumpFailure { it.copy(stageFailures = it.stageFailures + 1) } is WorkflowFailedEvent -> spanned.bumpFailure { it.copy(workflowFailures = it.workflowFailures + 1) } - else -> spanned + else -> spanned.applyQualityEvent(payload) } } + // Signal-quality events other bounded contexts emit. Split out of apply() so its when stays + // under the cyclomatic-complexity bar; unmatched payloads fall through unchanged. + private fun MetricsState.applyQualityEvent(payload: EventPayload): MetricsState = + when (payload) { + is SourceFetchedEvent -> bumpQuality { it.copy(sourceFetches = it.sourceFetches + 1) } + is LowQualityExtractionEvent -> + bumpQuality { it.copy(lowQualityExtractions = it.lowQualityExtractions + 1) } + is RepoKnowledgeRetrievedEvent -> + bumpQuality { it.copy(retrievedHits = it.retrievedHits + payload.hits.size) } + is RepoKnowledgeHitsFilteredEvent -> + bumpQuality { it.copy(droppedHits = it.droppedHits + payload.dropped.size) } + is BriefEchoMismatchEvent -> bumpQuality { it.copy(briefDriftCount = it.briefDriftCount + 1) } + is CapabilityGapReflectedEvent -> + if (payload.verdict == CapabilityGapVerdict.NEEDS_TOOL) { + bumpQuality { it.copy(capabilityGapCount = it.capabilityGapCount + 1) } + } else { + this + } + else -> this + } + private fun MetricsState.bumpFailure(transform: (FailureMetrics) -> FailureMetrics): MetricsState = copy(failures = transform(failures)) + private fun MetricsState.bumpQuality(transform: (QualityMetrics) -> QualityMetrics): MetricsState = + copy(quality = transform(quality)) + private fun MetricsState.applyInferenceCompleted(e: InferenceCompletedEvent): MetricsState { val provider = e.providerId.value val prev = inference.perProvider[provider] ?: ProviderInferenceMetrics() diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsState.kt b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsState.kt index 23b7175a..18dc7747 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsState.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/metrics/MetricsState.kt @@ -20,6 +20,7 @@ data class MetricsState( val tools: ToolMetrics = ToolMetrics(), val approvals: ApprovalMetrics = ApprovalMetrics(), val failures: FailureMetrics = FailureMetrics(), + val quality: QualityMetrics = QualityMetrics(), /** * In-flight approval requests awaiting resolution: requestId → request-event timestamp. * Transient correlation bookkeeping that lets request→decision latency be computed in a @@ -74,6 +75,21 @@ data class TierApprovalMetrics( val totalWaitMs: Long = 0, ) +/** + * Signal-quality accumulators — raw counts of the quality/rate events other bounded contexts emit. + * Ratios (extraction-quality, retrieval-precision) are derived at render time from these sums, the + * same way inference throughput is, so this stays a stable replayable accumulator. + */ +@Serializable +data class QualityMetrics( + val sourceFetches: Long = 0, // total research extractions (extraction-quality denominator) + val lowQualityExtractions: Long = 0, // extractions flagged below the quality bar + val retrievedHits: Long = 0, // repo-knowledge hits surfaced to a stage + val droppedHits: Long = 0, // repo-knowledge hits filtered below the similarity threshold + val briefDriftCount: Long = 0, // brief-echo mismatches (uncovered reqs / hallucinated refs) + val capabilityGapCount: Long = 0, // NEEDS_TOOL capability-gap reflections (genuine gaps only) +) + @Serializable data class FailureMetrics( val inferenceFailures: Long = 0, diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/narration/NarrationSubscriber.kt b/apps/server/src/main/kotlin/com/correx/apps/server/narration/NarrationSubscriber.kt index 46bb794f..4b3a9fe3 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/narration/NarrationSubscriber.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/narration/NarrationSubscriber.kt @@ -3,8 +3,11 @@ package com.correx.apps.server.narration import com.correx.core.events.events.ApprovalDecisionResolvedEvent import com.correx.core.events.events.ApprovalRequestedEvent import com.correx.core.events.events.ExecutionPlanRejectedEvent +import com.correx.core.events.events.InitialIntentEvent import com.correx.core.events.events.OrchestrationPausedEvent import com.correx.core.events.events.OrchestrationResumedEvent +import com.correx.core.events.events.ReviewFindingsRaisedEvent +import com.correx.core.events.events.ReviewVerdict import com.correx.core.events.events.StageCompletedEvent import com.correx.core.events.events.StageFailedEvent import com.correx.core.events.events.StoredEvent @@ -63,6 +66,16 @@ class NarrationSubscriber( val sid = event.metadata.sessionId when (val p = event.payload) { is WorkflowStartedEvent -> lanes[sid.value]?.let { it.used = 0 } + // Derive a human title from the opening intent, off the narration budget/lane (it's not + // a narration and must not consume a run's narration allowance). Launched so a slow name + // inference never stalls the event stream; failure is non-fatal (session keeps its id). + is InitialIntentEvent -> scope.launch { + runCatching { routerFacade.nameSession(sid, p.intent) } + .onFailure { e -> + if (e is CancellationException) throw e + log.warn("session naming failed for {}: {}", sid.value, e.message) + } + } // Approval pauses narrate off this event, not OrchestrationPaused: it carries the // tool/tier/preview facts directly, where the paused event forced a racy lookup // against state the next event had not yet written. @@ -107,6 +120,19 @@ class NarrationSubscriber( stageId = p.stageId.value, ), ) + // Surface the semantic reviewer's verdict conversationally instead of a raw findings + // dump: the narrator turns "FAIL, 2 findings" into a plain-language explanation the + // operator can act on. Only narrate when there's something to say (non-PASS or findings). + is ReviewFindingsRaisedEvent -> if (p.verdict != ReviewVerdict.PASS || p.findings.isNotEmpty()) { + enqueue( + sid, + NarrationTrigger( + kind = "review", + instruction = reviewInstruction(p), + stageId = p.stageId.value, + ), + ) + } is ExecutionPlanRejectedEvent -> enqueue( sid, NarrationTrigger( @@ -153,6 +179,25 @@ class NarrationSubscriber( } } + private fun reviewInstruction(event: ReviewFindingsRaisedEvent): String = buildString { + val verdict = event.verdict.name + append("The semantic reviewer returned $verdict for stage ${event.stageId.value}") + if (event.blocked) append(" (blocking — the stage will retry with this feedback)") + append(". ") + if (event.findings.isEmpty()) { + append("It raised no specific findings. Tell the operator the review result plainly.") + } else { + append("Summarise its findings for the operator in plain language:\n") + event.findings.take(MAX_REVIEW_FINDINGS).forEach { f -> + val loc = if (f.target.isNotBlank()) " [${f.target}]" else "" + appendLine("- ${f.severity} ${f.category}$loc: ${f.message.take(MAX_FINDING_CHARS)}") + } + if (event.findings.size > MAX_REVIEW_FINDINGS) { + appendLine("- …and ${event.findings.size - MAX_REVIEW_FINDINGS} more.") + } + } + } + private fun enqueue(sessionId: SessionId, trigger: NarrationTrigger, pauseKey: String? = null) { val lane = lanes.computeIfAbsent(sessionId.value) { startLane(sessionId) } if (lane.used >= maxPerRun) { @@ -211,6 +256,8 @@ class NarrationSubscriber( companion object { private const val DEFAULT_MAX_PER_RUN = 100 private const val MAX_PREVIEW_CHARS = 800 + private const val MAX_REVIEW_FINDINGS = 5 + private const val MAX_FINDING_CHARS = 200 /** Supersession scope for pauses with no approval request (e.g. USER_REQUESTED). */ private const val SESSION_PAUSE_KEY = "__session_pause__" diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/Dtos.kt b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/Dtos.kt index 3e26a697..8c43090f 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/Dtos.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/Dtos.kt @@ -84,6 +84,28 @@ data class EventEntryDto( val detail: String, ) +/** + * One raw OUTPUT-transcript source row for a reopened session. The server ships the *content* + * (chat text, narration, reasoning, tool diff/params/summary) with no formatting; the client folds + * these into its transcript rows in onSnapshot reusing its own diff/param/action formatters, so the + * row-rendering logic stays in one place (the Go client) and a restart restores the full transcript + * instead of only the thin EVENTS markers. [kind] is chat | narration | thinking | tool. + */ +@Serializable +data class TranscriptRowDto( + val kind: String, + val role: String = "", + val content: String = "", + val toolName: String = "", + val status: String = "", + val diff: String? = null, + val params: List = emptyList(), + val summary: String = "", + val affectedEntities: List = emptyList(), + val latencyMs: Long? = null, + val totalTokens: Int? = null, +) + @Serializable data class ToolRecordDto( val name: String, diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt index 94651bcd..bf218a4d 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/protocol/ServerMessage.kt @@ -51,6 +51,17 @@ sealed interface ServerMessage { override val sessionSequence: Long, ) : ServerMessage, SessionMessage + /** A human-readable title derived from the session's intent, from SessionNamedEvent — the + * client shows this in place of the opaque workflow id. */ + @Serializable + @SerialName("session.renamed") + data class SessionRenamed( + val sessionId: SessionId, + val name: String, + override val sequence: Long, + override val sessionSequence: Long, + ) : ServerMessage, SessionMessage + /** The workspace root a session was bound to, from SessionWorkspaceBoundEvent — shown as cwd. */ @Serializable @SerialName("session.workspace_bound") @@ -124,6 +135,9 @@ sealed interface ServerMessage { val pendingApprovals: List, val tools: List = emptyList(), val recentEvents: List = emptyList(), + // Raw source for the OUTPUT transcript (chat / narration / thinking / tool rows); the client + // folds these into its transcript on reopen so a restart restores the full conversation. + val transcript: List = emptyList(), // Last stage's inference output, resolved from CAS, so a reopened session restores its // output panel instead of showing blank until new activity arrives. val lastOutput: String = "", @@ -188,6 +202,10 @@ sealed interface ServerMessage { val stageId: StageId, val outputSummary: String, val responseText: String = "", + // The model's reasoning/thinking trace (llama `reasoning_content`), when the provider + // captured one. Sent so the client can offer a collapsed, opt-in "thinking" view; empty + // for models that don't emit a separate reasoning channel. + val reasoning: String = "", val occurredAt: Long, val totalTokens: Int? = null, override val sequence: Long, @@ -234,6 +252,10 @@ sealed interface ServerMessage { val sessionId: SessionId, val toolName: String, val tier: Tier, + // Pretty-formatted "key=value" call arguments (path=…, command="…", query=…) so the client + // can show what the tool was actually invoked with, not just its name. Capped + truncated + // upstream (see prettyToolParams) to keep the stream small. + val params: List = emptyList(), override val sequence: Long, override val sessionSequence: Long, ) : ServerMessage, SessionMessage diff --git a/apps/server/src/main/kotlin/com/correx/apps/server/routes/SessionRoutes.kt b/apps/server/src/main/kotlin/com/correx/apps/server/routes/SessionRoutes.kt index ee0e5705..512e42a8 100644 --- a/apps/server/src/main/kotlin/com/correx/apps/server/routes/SessionRoutes.kt +++ b/apps/server/src/main/kotlin/com/correx/apps/server/routes/SessionRoutes.kt @@ -52,6 +52,7 @@ data class StartSessionRequest( data class SessionSummaryResponse( val sessionId: String, val status: String, + val name: String? = null, val intent: String? = null, val workflowId: String? = null, val stageCount: Int = 0, @@ -84,6 +85,7 @@ fun Route.sessionRoutes(module: ServerModule) { SessionSummaryResponse( sessionId = s.sessionId.value, status = s.status, + name = s.name, intent = s.intent, workflowId = s.workflowId, stageCount = s.stageCount, diff --git a/apps/server/src/test/kotlin/com/correx/apps/server/bridge/DomainEventMapperTest.kt b/apps/server/src/test/kotlin/com/correx/apps/server/bridge/DomainEventMapperTest.kt index 5c9e43af..eb04f5b5 100644 --- a/apps/server/src/test/kotlin/com/correx/apps/server/bridge/DomainEventMapperTest.kt +++ b/apps/server/src/test/kotlin/com/correx/apps/server/bridge/DomainEventMapperTest.kt @@ -260,7 +260,7 @@ class DomainEventMapperTest { val result = domainEventToServerMessage(event, store, sessionSequence = 0L) assertEquals( ServerMessage.InferenceCompleted( - sessionId, stageId, responseText, responseText, occurredAt, + sessionId, stageId, responseText, responseText, occurredAt = occurredAt, totalTokens = 15, sequence = event.sequence, sessionSequence = 0L, ), result, @@ -283,7 +283,7 @@ class DomainEventMapperTest { val result = domainEventToServerMessage(event, noopStore, sessionSequence = 0L) assertEquals( ServerMessage.InferenceCompleted( - sessionId, stageId, "", "", occurredAt, + sessionId, stageId, "", "", occurredAt = occurredAt, totalTokens = 15, sequence = event.sequence, sessionSequence = 0L, ), result, @@ -325,7 +325,7 @@ class DomainEventMapperTest { ) val result = domainEventToServerMessage(event, noopStore, sessionSequence = 0L) assertEquals( - ServerMessage.ToolStarted(sessionId, "file_write", Tier.T2, event.sequence, 0L), + ServerMessage.ToolStarted(sessionId, "file_write", Tier.T2, sequence = event.sequence, sessionSequence = 0L), result, ) } diff --git a/apps/server/src/test/kotlin/com/correx/apps/server/bridge/SessionEventBridgeTest.kt b/apps/server/src/test/kotlin/com/correx/apps/server/bridge/SessionEventBridgeTest.kt index c6467745..3cd073cd 100644 --- a/apps/server/src/test/kotlin/com/correx/apps/server/bridge/SessionEventBridgeTest.kt +++ b/apps/server/src/test/kotlin/com/correx/apps/server/bridge/SessionEventBridgeTest.kt @@ -3,8 +3,15 @@ package com.correx.apps.server.bridge import com.correx.apps.server.protocol.EventEntryDto import com.correx.apps.server.protocol.ServerMessage import com.correx.apps.server.registry.WorkflowRegistry +import com.correx.core.approvals.Tier import com.correx.core.artifactstore.ArtifactStore +import com.correx.core.events.events.ChatTurnEvent +import com.correx.core.events.events.ChatTurnRole import com.correx.core.events.events.EventMetadata +import com.correx.core.events.events.InferenceCompletedEvent +import com.correx.core.events.events.TalkieNarrationEvent +import com.correx.core.events.events.ToolExecutionCompletedEvent +import com.correx.core.events.events.ToolReceipt import com.correx.core.events.events.ClarificationAnsweredEvent import com.correx.core.events.events.ClarificationAnswer import com.correx.core.events.events.ClarificationQuestion @@ -23,8 +30,12 @@ import com.correx.core.events.orchestration.OrchestrationStatus import com.correx.core.events.stores.EventStore import com.correx.core.events.types.ArtifactId import com.correx.core.events.types.ClarificationRequestId +import com.correx.core.inference.TokenUsage import com.correx.core.events.types.EventId +import com.correx.core.events.types.InferenceRequestId +import com.correx.core.events.types.ProviderId import com.correx.core.events.types.SessionId +import com.correx.core.events.types.ToolInvocationId import com.correx.core.events.types.StageId import com.correx.core.events.types.TransitionId import com.correx.core.kernel.orchestration.OrchestrationRepository @@ -228,6 +239,67 @@ class SessionEventBridgeTest { assertEquals(ServerMessage.SnapshotComplete, sent[1]) } + @Test + fun `replaySnapshot reconstructs the OUTPUT transcript with reasoning and a write row`() = runTest { + val reasoningId = ArtifactId("reasoning-1") + val artifactStore = object : ArtifactStore { + override suspend fun put(bytes: ByteArray): ArtifactId = ArtifactId("art-1") + override suspend fun get(id: ArtifactId): ByteArray? = + if (id == reasoningId) "thinking hard".toByteArray() else null + override suspend fun flushBefore(commit: suspend () -> Unit) = commit() + } + val diff = "--- a/app.css\n+++ b/app.css\n+body {}\n" + val receipt = ToolReceipt( + invocationId = ToolInvocationId("inv-1"), + toolName = "file_write", + exitCode = 0, + outputSummary = "wrote 1 line", + durationMs = 5L, + tier = Tier.T2, + timestamp = timestamp, + diff = diff, + ) + val events = listOf( + storedEvent(ChatTurnEvent(sessionId, "t1", ChatTurnRole.USER, "hi", timestampMs = 0L), seq = 1L), + storedEvent( + TalkieNarrationEvent(sessionId, "n1", "stage", stageId.value, "narrating", timestampMs = 0L), + seq = 2L, + ), + storedEvent( + InferenceCompletedEvent( + requestId = InferenceRequestId("r1"), + sessionId = sessionId, + stageId = stageId, + providerId = ProviderId("p"), + tokensUsed = TokenUsage(1, 1), + latencyMs = 3L, + responseArtifactId = ArtifactId("resp-1"), + reasoningArtifactId = reasoningId, + ), + seq = 3L, + ), + storedEvent( + ToolExecutionCompletedEvent(ToolInvocationId("inv-1"), sessionId, "file_write", receipt), + seq = 4L, + ), + ) + val store = fakeEventStore(allEventsList = events) + val sent = mutableListOf() + val bridge = SessionEventBridge( + store, artifactStore, activeOrchestrationRepository(), + noopWorkflowRegistry, noopToolRegistry, + ) { sent.add(it) } + + bridge.replaySnapshot() + + val transcript = (sent[0] as ServerMessage.SessionSnapshot).transcript + assertEquals(listOf("chat", "narration", "thinking", "tool"), transcript.map { it.kind }) + assertEquals("USER", transcript[0].role) + assertEquals("thinking hard", transcript[2].content) + assertEquals(diff, transcript[3].diff) + assertEquals("wrote 1 line", transcript[3].summary) + } + @Test fun `replaySnapshot populates recentEvents from event store`() = runTest { val events = listOf( diff --git a/apps/server/src/test/kotlin/com/correx/apps/server/metrics/MetricsProjectionTest.kt b/apps/server/src/test/kotlin/com/correx/apps/server/metrics/MetricsProjectionTest.kt index edc09837..b619e3b8 100644 --- a/apps/server/src/test/kotlin/com/correx/apps/server/metrics/MetricsProjectionTest.kt +++ b/apps/server/src/test/kotlin/com/correx/apps/server/metrics/MetricsProjectionTest.kt @@ -5,11 +5,19 @@ import com.correx.core.approvals.ApprovalStatus import com.correx.core.approvals.Tier import com.correx.core.events.events.ApprovalDecisionResolvedEvent import com.correx.core.events.events.ApprovalRequestedEvent +import com.correx.core.events.events.BriefEchoMismatchEvent +import com.correx.core.events.events.CapabilityGapReflectedEvent +import com.correx.core.events.events.CapabilityGapVerdict import com.correx.core.events.events.EventMetadata import com.correx.core.events.events.EventPayload import com.correx.core.events.events.InferenceCompletedEvent import com.correx.core.events.events.InferenceFailedEvent +import com.correx.core.events.events.LowQualityExtractionEvent import com.correx.core.events.events.NewEvent +import com.correx.core.events.events.RepoKnowledgeHit +import com.correx.core.events.events.RepoKnowledgeHitsFilteredEvent +import com.correx.core.events.events.RepoKnowledgeRetrievedEvent +import com.correx.core.events.events.SourceFetchedEvent import com.correx.core.events.events.StoredEvent import com.correx.core.events.events.ToolExecutionCompletedEvent import com.correx.core.events.events.ToolExecutionFailedEvent @@ -254,4 +262,43 @@ class MetricsProjectionTest { assertEquals(0.0, r.tokensPerSecond, 0.001) assertEquals(0.0, r.inferencePct, 0.001) } + + @Test + fun `quality events derive extraction quality, retrieval precision, and drift-gap counts`() { + val hit = { p: String -> RepoKnowledgeHit(path = p, text = "t", score = 0.7f) } + val qualitySeed = listOf( + stored(SourceFetchedEvent(sessionId, stageId, "u1", "sha1", quality = "ok"), 1L, at(1)), + stored(SourceFetchedEvent(sessionId, stageId, "u2", "sha2", quality = "ok"), 2L, at(2)), + stored(SourceFetchedEvent(sessionId, stageId, "u3", "sha3", quality = "low"), 3L, at(3)), + stored(LowQualityExtractionEvent(sessionId, stageId, "u3", "sha3", "paywall"), 4L, at(4)), + stored( + RepoKnowledgeRetrievedEvent(sessionId, stageId, "q", listOf(hit("a"), hit("b"), hit("c"))), + 5L, at(5), + ), + stored( + RepoKnowledgeHitsFilteredEvent(sessionId, "q", threshold = 0.5f, dropped = listOf(hit("d"))), + 6L, at(6), + ), + stored(BriefEchoMismatchEvent(sessionId, stageId, listOf("req1"), emptyList(), listOf("Foo")), 7L, at(7)), + stored( + CapabilityGapReflectedEvent(sessionId, stageId, "file_edit", CapabilityGapVerdict.NEEDS_TOOL, "d", 0L), + 8L, at(8), + ), + // RESOLVED = self-corrected, not a genuine gap — must not count. + stored( + CapabilityGapReflectedEvent(sessionId, stageId, "file_write", CapabilityGapVerdict.RESOLVED, "d", 0L), + 9L, at(9), + ), + ) + val q = MetricsInspectionService(SeededEventStore(qualitySeed)).inspect(sessionId).quality + + assertEquals(3L, q.sourceFetches) + assertEquals(1L, q.lowQualityExtractions) + assertEquals(2.0 / 3.0, q.extractionQualityRate, 0.001) + assertEquals(3L, q.retrievedHits) + assertEquals(1L, q.droppedHits) + assertEquals(0.75, q.retrievalPrecision, 0.001) + assertEquals(1L, q.briefDriftCount) + assertEquals(1L, q.capabilityGapCount) + } } diff --git a/apps/server/src/test/kotlin/com/correx/apps/server/narration/NarrationSubscriberTest.kt b/apps/server/src/test/kotlin/com/correx/apps/server/narration/NarrationSubscriberTest.kt index b946dfdf..a82cd428 100644 --- a/apps/server/src/test/kotlin/com/correx/apps/server/narration/NarrationSubscriberTest.kt +++ b/apps/server/src/test/kotlin/com/correx/apps/server/narration/NarrationSubscriberTest.kt @@ -81,6 +81,8 @@ class NarrationSubscriberTest { override suspend fun narrate(sessionId: SessionId, trigger: NarrationTrigger) { calls.add(sessionId to trigger) } + + override suspend fun nameSession(sessionId: SessionId, intent: String) = Unit } private fun storedEvent(payload: EventPayload, seq: Long): StoredEvent = StoredEvent( @@ -191,6 +193,8 @@ class NarrationSubscriberTest { if (callCount == 1) throw RuntimeException("simulated failure") calls.add(trigger.kind) } + + override suspend fun nameSession(sessionId: SessionId, intent: String) = Unit } NarrationSubscriber(fakeStore, facade, scope).start() @@ -263,6 +267,8 @@ class NarrationSubscriberTest { } calls.add(trigger) } + + override suspend fun nameSession(sessionId: SessionId, intent: String) = Unit } private fun approvalRequested(requestId: String): ApprovalRequestedEvent = ApprovalRequestedEvent( diff --git a/apps/tui-go/internal/app/model.go b/apps/tui-go/internal/app/model.go index f54b679e..2fc9b941 100644 --- a/apps/tui-go/internal/app/model.go +++ b/apps/tui-go/internal/app/model.go @@ -119,6 +119,7 @@ type ToolRecord struct { Name string Tier int Status ToolStatus + Params []string // pretty "key=value" call args, from tool.started } // ManifestTool is a declared (not-yet-run) tool from a stage manifest. @@ -208,6 +209,7 @@ type Session struct { Status string WorkflowID string Name string + named bool // Name came from a session.renamed (intent-derived title); don't clobber with workflowId LastEventAt int64 CurrentStage string PlanGoal string @@ -429,6 +431,11 @@ type Model struct { // the OUTPUT transcript; they always stay in the EVENTS panel. Persisted in tui-prefs.json. actionsHidden bool + // thinkingShown reveals the model's reasoning/thinking blocks in the OUTPUT transcript. + // Off by default (collapsed to a one-line summary) so the trace doesn't drown the answer; + // toggled from the palette ("thinking"). Persisted in tui-prefs.json. + thinkingShown bool + // outputScroll is how many rows up from the bottom the OUTPUT transcript is scrolled // (0 = tail-follow the newest output). PgUp/PgDn + ctrl+u/d move it; esc snaps back. outputScroll int @@ -504,6 +511,7 @@ func NewModel(client *ws.Client) Model { transcriptSel: -1, sbHidden: loadStatusbarHidden(), actionsHidden: loadPrefs().InlineActionsHidden, + thinkingShown: loadPrefs().ThinkingShown, } } diff --git a/apps/tui-go/internal/app/overlays.go b/apps/tui-go/internal/app/overlays.go index 5d2c9981..41b0678b 100644 --- a/apps/tui-go/internal/app/overlays.go +++ b/apps/tui-go/internal/app/overlays.go @@ -532,7 +532,7 @@ func (m Model) eventInspectorModal() string { fg = t.P.FgStrong } row := marker + mbg(t, padRaw(e.Time, 8)+" ", t.P.Faint) + - lipgloss.NewStyle().Foreground(t.categoryColor(cat)).Background(t.P.BgPanel).Render(padRaw("["+cat+"]", 11)) + " " + + lipgloss.NewStyle().Foreground(t.categoryColor(cat)).Background(t.P.BgPanel).Render(padRaw("["+cat+"]", 11)) + mbg(t, " ", t.P.Fg) + lipgloss.NewStyle().Foreground(fg).Background(t.P.BgPanel).Render(padRaw(e.Type, 18)) + mbg(t, " "+truncate(e.Detail, detailBudget), t.P.Dim) b.WriteString(row + "\n") diff --git a/apps/tui-go/internal/app/prefs.go b/apps/tui-go/internal/app/prefs.go index c6727065..027987ae 100644 --- a/apps/tui-go/internal/app/prefs.go +++ b/apps/tui-go/internal/app/prefs.go @@ -11,6 +11,7 @@ import ( type prefs struct { StatusbarHidden []string `json:"statusbarHidden"` InlineActionsHidden bool `json:"inlineActionsHidden"` + ThinkingShown bool `json:"thinkingShown"` } // statusSegment is one toggleable status-bar segment. The always-on segments (correx label, @@ -97,3 +98,10 @@ func saveInlineActionsHidden(hidden bool) { p.InlineActionsHidden = hidden savePrefs(p) } + +// saveThinkingShown persists the reasoning-blocks toggle, preserving other prefs fields. +func saveThinkingShown(shown bool) { + p := loadPrefs() + p.ThinkingShown = shown + savePrefs(p) +} diff --git a/apps/tui-go/internal/app/render_matrix_test.go b/apps/tui-go/internal/app/render_matrix_test.go index 486ed754..996e2a39 100644 --- a/apps/tui-go/internal/app/render_matrix_test.go +++ b/apps/tui-go/internal/app/render_matrix_test.go @@ -352,6 +352,17 @@ func matrixCases() []matrixCase { want: []string{"corx-ws"}, }, + // --- session rename (status bar shows intent-derived title) --- + { + name: "session.renamed", + kinds: []string{protocol.TypeSessionRenamed}, + surface: surfaceStatusBar, + build: func() protocol.ServerMessage { + return msg(protocol.TypeSessionRenamed, func(m *protocol.ServerMessage) { m.Name = "Add Auth Layer" }) + }, + want: []string{"Add Auth Layer"}, + }, + // --- approval gate (docked band card) --- { name: "approval.required", diff --git a/apps/tui-go/internal/app/server.go b/apps/tui-go/internal/app/server.go index 0e97fe3d..a6ae864b 100644 --- a/apps/tui-go/internal/app/server.go +++ b/apps/tui-go/internal/app/server.go @@ -96,6 +96,8 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { switch msg.Type { case protocol.TypeSessionAnnounced: m.onSessionAnnounced(msg) + case protocol.TypeSessionRenamed: + m.onSessionRenamed(msg) case protocol.TypeRouterNarration: entry := RouterEntry{Role: "narration_llm", Content: msg.Content} if msg.LatencyMs != nil && msg.TotalTokens != nil { @@ -198,6 +200,11 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { s.StageTokensUsed = *msg.TotalTokens } s.addEvent(msg.OccurredAt, "InferenceCompleted", msg.StageID) + // Thread the model's reasoning trace into the transcript as a collapsed "thinking" + // row (revealed via the palette). Skipped when the model emits no separate channel. + if strings.TrimSpace(msg.Reasoning) != "" { + m.appendRouter(msg.SessionID, RouterEntry{Role: "thinking", Content: msg.Reasoning}) + } } case protocol.TypeInferenceTimeout: if s := m.session(msg.SessionID); s != nil { @@ -217,7 +224,7 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { case protocol.TypeToolStarted: if s := m.session(msg.SessionID); s != nil { s.Active = true - s.Tools = append(s.Tools, ToolRecord{Name: msg.ToolName, Status: ToolStarted}) + s.Tools = append(s.Tools, ToolRecord{Name: msg.ToolName, Status: ToolStarted, Params: msg.Params}) if len(s.Tools) > 8 { s.Tools = s.Tools[len(s.Tools)-8:] } @@ -240,7 +247,15 @@ func (m *Model) applyServer(msg protocol.ServerMessage) { m.appendRouter(msg.SessionID, RouterEntry{Role: "tool", Content: *msg.Diff}) } else { label := msg.ToolName - if len(msg.AffectedEntities) > 0 { + // Prefer the actual call args (path=…, command="…") over the affected-entities + // fallback: params show *what was asked*, not just what got touched. + if s := m.session(msg.SessionID); s != nil { + if suffix := paramSuffix(lastToolParams(s, msg.ToolName)); suffix != "" { + label += suffix + } else if len(msg.AffectedEntities) > 0 { + label += " " + strings.Join(msg.AffectedEntities, ", ") + } + } else if len(msg.AffectedEntities) > 0 { label += " " + strings.Join(msg.AffectedEntities, ", ") } m.appendAction(msg.SessionID, "✓", actionToolText(label, msg.Summary)) @@ -542,6 +557,25 @@ func countSuffix(added, removed int) string { return " (+" + itoa(added) + " −" + itoa(removed) + ")" } +// lastToolParams returns the call args of the most recent invocation of the named tool. The ReAct +// loop runs one tool at a time per session, so the last started record is the one now completing. +func lastToolParams(s *Session, name string) []string { + for i := len(s.Tools) - 1; i >= 0; i-- { + if s.Tools[i].Name == name { + return s.Tools[i].Params + } + } + return nil +} + +// paramSuffix renders pretty call args as a " (k=v · k=v)" suffix, or "" when there are none. +func paramSuffix(params []string) string { + if len(params) == 0 { + return "" + } + return " (" + clip(strings.Join(params, " · "), 56) + ")" +} + // actionToolText joins a tool label with a short, clipped result summary. func actionToolText(label, summary string) string { // Collapse to a single line first: tool summaries (dir listings, file heads) @@ -589,7 +623,9 @@ func (m *Model) onSessionAnnounced(msg protocol.ServerMessage) { s.Status = "ACTIVE" if msg.WorkflowID != "" { s.WorkflowID = msg.WorkflowID - s.Name = msg.WorkflowID + if !s.named { + s.Name = msg.WorkflowID + } } s.LastEventAt = nowMillis() if m.pendingWorkflowFocus { @@ -601,6 +637,19 @@ func (m *Model) onSessionAnnounced(msg protocol.ServerMessage) { } } +// onSessionRenamed applies the intent-derived title from session.renamed, replacing the +// opaque workflow id shown until the naming inference completed. The named flag pins it so a +// later announce (e.g. on reconnect replay) doesn't revert the display back to the workflow id. +func (m *Model) onSessionRenamed(msg protocol.ServerMessage) { + if msg.Name == "" { + return + } + s := m.ensureSession(msg.SessionID) + s.Name = msg.Name + s.named = true + s.LastEventAt = nowMillis() +} + func (m *Model) onApprovalRequired(msg protocol.ServerMessage) { risk := "unknown" var rationale []string @@ -710,6 +759,70 @@ func (m *Model) onSnapshot(msg protocol.ServerMessage) { if m.selectedID == "" { m.selectedID = msg.SessionID } + m.restoreTranscript(msg.SessionID, msg.Transcript) +} + +// restoreTranscript rebuilds the OUTPUT transcript for a reopened session from the snapshot's raw +// source rows, reusing the same formatters the live path uses (diffSummary/paramSuffix/…). Replaces +// (not appends) so a reconnect with an existing transcript doesn't duplicate rows. +func (m *Model) restoreTranscript(sid string, rows []protocol.TranscriptRowDto) { + if len(rows) == 0 { + return + } + out := make([]RouterEntry, 0, len(rows)) + for _, r := range rows { + switch r.Kind { + case "chat": + role := "router" + if r.Role == "USER" { + role = "user" + } + e := RouterEntry{Role: role, Content: r.Content} + if r.LatencyMs != nil && r.TotalTokens != nil { + e.Metrics = &TurnMetrics{LatencyMs: *r.LatencyMs, TotalTokens: *r.TotalTokens} + } + out = append(out, e) + case "narration": + e := RouterEntry{Role: "narration_llm", Content: r.Content} + if r.LatencyMs != nil && r.TotalTokens != nil { + e.Metrics = &TurnMetrics{LatencyMs: *r.LatencyMs, TotalTokens: *r.TotalTokens} + } + out = append(out, e) + case "thinking": + if strings.TrimSpace(r.Content) != "" { + out = append(out, RouterEntry{Role: "thinking", Content: r.Content}) + } + case "tool": + out = append(out, toolTranscriptRows(r)...) + } + } + m.routerMessages[sid] = out +} + +// toolTranscriptRows mirrors the live TypeToolCompleted/Failed/Rejected handling: a diff becomes a +// ✎ write row + the collapsed raw-diff row; otherwise a single ✓/✗/✕ action row. +func toolTranscriptRows(r protocol.TranscriptRowDto) []RouterEntry { + if r.Diff != nil && *r.Diff != "" { + path, add, del := diffSummary(*r.Diff) + return []RouterEntry{ + {Role: "action", Icon: "✎", Content: "wrote " + path + countSuffix(add, del)}, + {Role: "tool", Content: *r.Diff}, + } + } + switch r.Status { + case "failed": + return []RouterEntry{{Role: "action", Icon: "✗", Content: actionToolText(r.ToolName+" failed", r.Summary)}} + case "rejected": + return []RouterEntry{{Role: "action", Icon: "✕", Content: actionToolText(r.ToolName+" blocked", r.Summary)}} + default: + label := r.ToolName + if suffix := paramSuffix(r.Params); suffix != "" { + label += suffix + } else if len(r.AffectedEntities) > 0 { + label += " " + strings.Join(r.AffectedEntities, ", ") + } + return []RouterEntry{{Role: "action", Icon: "✓", Content: actionToolText(label, r.Summary)}} + } } func (m *Model) touch(id, status string) { diff --git a/apps/tui-go/internal/app/sessions_overlay.go b/apps/tui-go/internal/app/sessions_overlay.go index 4963ad1b..b4198831 100644 --- a/apps/tui-go/internal/app/sessions_overlay.go +++ b/apps/tui-go/internal/app/sessions_overlay.go @@ -16,6 +16,7 @@ import ( type SessionSummary struct { SessionID string `json:"sessionId"` Status string `json:"status"` + Name string `json:"name"` Intent string `json:"intent"` WorkflowID string `json:"workflowId"` StageCount int `json:"stageCount"` @@ -222,7 +223,12 @@ func (m Model) sessionListRow(i int) string { statusCell := lipgloss.NewStyle().Foreground(statusColor(t, s.Status)).Background(t.P.BgPanel).Bold(true). Render(padRaw(statusLabel(s.Status), 9)) - stage := s.WorkflowID + // Prefer the human session name (SessionNamedEvent, intent-derived) over the raw + // workflow id — it's what the operator recognizes a session by. + stage := s.Name + if stage == "" { + stage = s.WorkflowID + } if s.StageCount > 0 { stage += " ·" + itoa(s.StageCount) + "stg" } diff --git a/apps/tui-go/internal/app/update.go b/apps/tui-go/internal/app/update.go index 550e53fe..9ca01984 100644 --- a/apps/tui-go/internal/app/update.go +++ b/apps/tui-go/internal/app/update.go @@ -1160,6 +1160,7 @@ func paletteCommands() []paletteCmd { {"rail", "", "idle rail", "show / hide the idle status + keys rail"}, {"panel", "d", "right panel", "in-session: events → changes → off"}, {"actions", "", "inline actions", "show / hide tool & action rows in output"}, + {"thinking", "", "thinking blocks", "show / hide model reasoning in output"}, {"help", "?", "help", "keybinding cheat-sheet"}, {"mode", "s", "toggle mode", "switch chat / steering"}, {"cancel", "c", "cancel session", "stop the selected session"}, @@ -1217,6 +1218,9 @@ func (m Model) execPalette(id string) (tea.Model, tea.Cmd) { m.cycleRightPanel() case "actions": m.toggleInlineActions() + case "thinking": + m.thinkingShown = !m.thinkingShown + saveThinkingShown(m.thinkingShown) case "help": m.overlay = OverlayHelp m.modalScroll = 0 diff --git a/apps/tui-go/internal/app/view.go b/apps/tui-go/internal/app/view.go index 3595089b..84eaecf4 100644 --- a/apps/tui-go/internal/app/view.go +++ b/apps/tui-go/internal/app/view.go @@ -573,7 +573,7 @@ func (m Model) launcherInput(w int) string { box := strings.Join(boxLines, "\n") wf := lipgloss.NewStyle().Foreground(t.P.Accent2).Background(t.P.Bg).Render(m.launcherWfName()) - leftSub := " " + wf + t.span(" · ", t.P.Faint) + t.span(m.currentModel, t.P.Faint) + leftSub := t.span(" ", t.P.Faint) + wf + t.span(" · ", t.P.Faint) + t.span(m.currentModel, t.P.Faint) hints := t.span("Tab wf · ⌥↵/^J newline · enter ↵ ", t.P.Faint) return box + "\n" + m.justify(leftSub, hints, w, t.P.Bg) } @@ -823,6 +823,23 @@ func (m Model) buildTranscriptRows(w int) ([]string, []int) { } case "tool": rows = append(rows, t.span(toolRowSummary(e.Content), t.P.Dim)) + case "thinking": + // Collapsed by default to a single muted line so the reasoning trace doesn't bury + // the answer; the palette "thinking" toggle reveals the full dimmed block. + brain := lipgloss.NewStyle().Foreground(t.P.Faint).Background(t.P.Bg).Render("✼") + if !m.thinkingShown { + n := strings.Count(strings.TrimRight(e.Content, "\n"), "\n") + 1 + rows = append(rows, brain+t.span(" thinking ("+plural(n, "line")+") — palette: thinking", t.P.Faint)) + break + } + lines := wrap(e.Content, w-2) + for i, ln := range lines { + if i == 0 { + rows = append(rows, brain+t.span(" ", t.P.Bg)+t.span(ln, t.P.Faint)) + } else { + rows = append(rows, t.span(" ", t.P.Bg)+t.span(ln, t.P.Faint)) + } + } case "narration_llm": diamond := lipgloss.NewStyle().Foreground(t.P.Accent).Background(t.P.Bg).Bold(true).Render("◆") lines := wrap(e.Content, w-2) @@ -913,7 +930,7 @@ func (m Model) eventRows(w, h int) []string { catCell := lipgloss.NewStyle().Foreground(t.categoryColor(cat)).Background(t.P.BgPanel). Render(" " + padRaw(cat, 9) + " ") evRows = append(evRows, - t.span(e.Time+" ", t.P.Faint)+catCell+" "+ + t.span(e.Time+" ", t.P.Faint)+catCell+t.span(" ", t.P.Bg)+ t.span(padRaw(e.Type, 18), t.P.Fg)+t.span(e.Detail, t.P.Dim)) } // Show the latest events that fit under the 2-line header (box inner = h-2). diff --git a/apps/tui-go/internal/protocol/protocol.go b/apps/tui-go/internal/protocol/protocol.go index 8e1b8794..da619e66 100644 --- a/apps/tui-go/internal/protocol/protocol.go +++ b/apps/tui-go/internal/protocol/protocol.go @@ -36,6 +36,7 @@ const ( TypeApprovalResolved = "approval.resolved" TypeClarification = "clarification.required" TypeWorkflowProposed = "workflow.proposed" + TypeSessionRenamed = "session.renamed" TypeWorkspaceBound = "session.workspace_bound" TypeStageManifest = "stage.tool_manifest" TypeProviderStatus = "provider.status_changed" @@ -70,30 +71,34 @@ const ( type ServerMessage struct { Type string `json:"type"` - SessionID string `json:"sessionId"` - WorkflowID string `json:"workflowId"` - StageID string `json:"stageId"` - ToolName string `json:"toolName"` - Role string `json:"role"` - TurnID string `json:"turnId"` - Reason string `json:"reason"` - Summary string `json:"outputSummary"` - Response string `json:"responseText"` - LastOutput string `json:"lastOutput"` // session_snapshot: last stage output (reopen) - LastResponse string `json:"lastResponse"` // session_snapshot: last stage response (reopen) - Content string `json:"content"` - Diff *string `json:"diff"` + SessionID string `json:"sessionId"` + WorkflowID string `json:"workflowId"` + StageID string `json:"stageId"` + ToolName string `json:"toolName"` + Role string `json:"role"` + TurnID string `json:"turnId"` + Reason string `json:"reason"` + Summary string `json:"outputSummary"` + Response string `json:"responseText"` + Reasoning string `json:"reasoning"` // inference.completed: model thinking trace (collapsed view) + LastOutput string `json:"lastOutput"` // session_snapshot: last stage output (reopen) + LastResponse string `json:"lastResponse"` // session_snapshot: last stage response (reopen) + Content string `json:"content"` + Diff *string `json:"diff"` // tool.completed: file/dir path(s) the tool read, wrote, or listed. AffectedEntities []string `json:"affectedEntities"` - Tier string `json:"tier"` - Message string `json:"message"` - RequestID string `json:"requestId"` - ArtifactID string `json:"artifactId"` - ProviderID string `json:"providerId"` - Preview *string `json:"preview"` - Disposition string `json:"disposition"` - Outcome string `json:"outcome"` // approval.resolved: APPROVED | REJECTED | AUTO_APPROVED - WorkspaceRoot string `json:"workspaceRoot"` // session.workspace_bound: the bound cwd + // tool.started: pretty "key=value" call arguments (path=…, command="…"). + Params []string `json:"params"` + Tier string `json:"tier"` + Message string `json:"message"` + RequestID string `json:"requestId"` + ArtifactID string `json:"artifactId"` + ProviderID string `json:"providerId"` + Preview *string `json:"preview"` + Disposition string `json:"disposition"` + Outcome string `json:"outcome"` // approval.resolved: APPROVED | REJECTED | AUTO_APPROVED + WorkspaceRoot string `json:"workspaceRoot"` // session.workspace_bound: the bound cwd + Name string `json:"name"` // session.renamed: human title derived from intent // plan.locked: the compiled phase-2 stage ids in execution order StageIDs []string `json:"stageIds"` @@ -134,6 +139,7 @@ type ServerMessage struct { PendingAppr []ApprovalDto `json:"pendingApprovals"` Tools []ToolRecordDto `json:"tools"` RecentEvents []EventEntryDto `json:"recentEvents"` + Transcript []TranscriptRowDto `json:"transcript"` Stages []StageToolDecl `json:"stages"` Workflows []WorkflowDto `json:"workflows"` AssessedIssues []AssessedIssueDto `json:"issues"` @@ -387,6 +393,22 @@ type ToolRecordDto struct { Diff *string `json:"diff"` } +// TranscriptRowDto is one raw OUTPUT-transcript source row from a session_snapshot; the client +// folds these into RouterEntry rows on reopen, applying its own diff/param/action formatting. +type TranscriptRowDto struct { + Kind string `json:"kind"` + Role string `json:"role"` + Content string `json:"content"` + ToolName string `json:"toolName"` + Status string `json:"status"` + Diff *string `json:"diff"` + Params []string `json:"params"` + Summary string `json:"summary"` + AffectedEntities []string `json:"affectedEntities"` + LatencyMs *int64 `json:"latencyMs"` + TotalTokens *int `json:"totalTokens"` +} + type WorkflowDto struct { WorkflowID string `json:"workflowId"` Description string `json:"description"` @@ -413,7 +435,7 @@ func (m ServerMessage) IsEventBearing() bool { TypeToolAssessed, TypeApprovalRequired, TypeApprovalResolved, TypeClarification, TypeWorkflowProposed, - TypeWorkspaceBound, TypeArtifactCreated, + TypeSessionRenamed, TypeWorkspaceBound, TypeArtifactCreated, TypeArtifactValid, TypePlanLocked, TypeRouterNarration, TypeReviewFindings: return true diff --git a/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt b/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt index 03499627..23126f28 100644 --- a/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt +++ b/core/context/src/main/kotlin/com/correx/core/context/builder/DefaultContextPackBuilder.kt @@ -54,7 +54,10 @@ class DefaultContextPackBuilder( // decisionJournal is deliberately NOT here anymore: the journal is OPTIONAL-bucket, capped // by OptionalCeilings. Pinning is otherwise bucket-driven (ContextBucket.REQUIRED); this set // only covers entries built outside the orchestrator that never stamp buckets. - private val neverDropSourceTypes = setOf("steeringNote", "eventHistory", "factSheet") + // "remainingDelta" is the shrinking stage-contract checklist (stage-termination design + // 2026-07-11): it must survive every budget/dedup pass, since its entire purpose is to give + // the model a progress signal that outlives the truncation which otherwise wipes its memory. + private val neverDropSourceTypes = setOf("steeringNote", "eventHistory", "factSheet", "remainingDelta") private companion object { const val CHARS_PER_TOKEN = 4 diff --git a/core/context/src/test/kotlin/RemainingDeltaPinningTest.kt b/core/context/src/test/kotlin/RemainingDeltaPinningTest.kt new file mode 100644 index 00000000..956f63b9 --- /dev/null +++ b/core/context/src/test/kotlin/RemainingDeltaPinningTest.kt @@ -0,0 +1,52 @@ +import com.correx.core.context.builder.DefaultContextPackBuilder +import com.correx.core.context.compression.DefaultContextCompressor +import com.correx.core.context.model.ContextEntry +import com.correx.core.context.model.ContextLayer +import com.correx.core.context.model.TokenBudget +import com.correx.core.events.types.ContextEntryId +import com.correx.core.events.types.ContextPackId +import com.correx.core.events.types.SessionId +import com.correx.core.events.types.StageId +import org.junit.jupiter.api.Assertions.assertTrue +import org.junit.jupiter.api.Test + +/** + * The remaining-delta checklist (stage-termination design 2026-07-11) must survive the budget/dedup + * passes — its whole point is to outlive the truncation that gives the model amnesia. It is pinned + * via sourceType "remainingDelta" in [DefaultContextPackBuilder.neverDropSourceTypes]. + */ +class RemainingDeltaPinningTest { + + private val builder = DefaultContextPackBuilder(DefaultContextCompressor()) + private val sessionId = SessionId("s1") + private val stageId = StageId("stage-1") + private val packId = ContextPackId("pack-1") + + @Test + fun `remainingDelta entry survives a wildly over-budget build`() = kotlinx.coroutines.runBlocking { + val checklist = ContextEntry( + id = ContextEntryId("delta-1"), + layer = ContextLayer.L1, + content = "## Remaining to finish this stage\n- [ ] Foo.tsx — exports_default_component", + sourceType = "remainingDelta", + sourceId = "remaining-delta", + tokenEstimate = 500, + ) + // A large freeform entry that would be dropped to fit the tiny budget. + val filler = ContextEntry( + id = ContextEntryId("filler-1"), + layer = ContextLayer.L2, + content = "lorem ipsum ".repeat(200), + sourceType = "toolResult", + sourceId = "filler-1", + tokenEstimate = 5000, + ) + val pack = builder.build(packId, sessionId, stageId, listOf(filler, checklist), TokenBudget(limit = 10)) + val retained = pack.layers.values.flatten() + assertTrue( + retained.any { it.sourceType == "remainingDelta" }, + "remainingDelta checklist must be retained even when the budget is blown", + ) + Unit + } +} diff --git a/core/events/src/main/kotlin/com/correx/core/events/events/IntentEvents.kt b/core/events/src/main/kotlin/com/correx/core/events/events/IntentEvents.kt index ee5e8552..84388502 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/events/IntentEvents.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/events/IntentEvents.kt @@ -16,3 +16,18 @@ data class InitialIntentEvent( val sessionId: SessionId, val intent: String, ) : EventPayload + +/** + * A short, human-readable title for the session (e.g. "Add auth to account service"), derived + * once from the initial intent by the Talkie model. Recorded as an event (invariant #9) so the + * nondeterministic name is generated a single time and replays identically — the TUI shows this + * instead of the opaque workflow id. Non-authoritative display metadata; nothing in the core + * decision path reads it. + */ +@Serializable +@SerialName("SessionNamed") +data class SessionNamedEvent( + val sessionId: SessionId, + val name: String, + val timestampMs: Long, +) : EventPayload diff --git a/core/events/src/main/kotlin/com/correx/core/events/events/OrchestrationEvents.kt b/core/events/src/main/kotlin/com/correx/core/events/events/OrchestrationEvents.kt index 99c1c7d7..9f9e924c 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/events/OrchestrationEvents.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/events/OrchestrationEvents.kt @@ -36,6 +36,23 @@ data class WorkflowFailedEvent( val retryExhausted: Boolean, ) : EventPayload +/** + * Records that the operator approved access to a specific path OUTSIDE the workspace root + * (a `file_read`/`list_dir` target). The intent plane raises PROMPT_USER for any out-of-workspace + * read; on approval this event is recorded so (a) subsequent reads of the same path this session + * skip the prompt, and (b) the tool jail is widened to admit it. [path] is the lexically-resolved + * absolute path — the recorded fact replay reads instead of re-prompting (invariants #8/#9). + * Scope is the session: nothing here persists across sessions. Privileged locations are BLOCKED + * upstream and never reach this event. + */ +@Serializable +@SerialName("OutsidePathAccessGranted") +data class OutsidePathAccessGrantedEvent( + val sessionId: SessionId, + val stageId: StageId, + val path: String, +) : EventPayload + @Serializable @SerialName("OrchestrationPaused") data class OrchestrationPausedEvent( @@ -115,6 +132,12 @@ data class FailureTicketOpenedEvent( // Fingerprint of [evidence] (whitespace/digit-normalised, see FailureFingerprint) at route time. // The next route compares against this to decide progress vs. no-progress. Empty on old events. val fingerprint: String = "", + // Tier-2 escalation marker. False = tier-1, the ticket routed to the file's own author to patch its + // own layer. True = the author loop exhausted its budget, so the ticket escalated to the arbiter + // (the recovery stage recast as intent-holder): the failure is a cross-file CONTRACT dispute no + // single owner can settle, so the arbiter is given the initial intent as authority and told to + // reconcile ALL the named files in one pass. Defaulted for backward-compatible replay. + val escalated: Boolean = false, ) : EventPayload /** diff --git a/core/events/src/main/kotlin/com/correx/core/events/events/ToolRequest.kt b/core/events/src/main/kotlin/com/correx/core/events/events/ToolRequest.kt index 0f5ec66b..9f50a720 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/events/ToolRequest.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/events/ToolRequest.kt @@ -13,5 +13,9 @@ data class ToolRequest( val stageId: StageId, val toolName: String, @Serializable(with = AnyMapSerializer::class) - val parameters: Map = emptyMap() + val parameters: Map = emptyMap(), + // Paths outside the workspace root that the operator has approved for this call (recorded via + // OutsidePathAccessGrantedEvent). Read tools widen their containment jail to admit these; write + // tools ignore them, so writes stay workspace-jailed. Empty by default = pure workspace jail. + val grantedPaths: Set = emptySet(), ) diff --git a/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt b/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt index 5f282021..51a940fe 100644 --- a/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt +++ b/core/events/src/main/kotlin/com/correx/core/events/serialization/Serialization.kt @@ -49,6 +49,7 @@ import com.correx.core.events.events.L3MemoryRetrievedEvent import com.correx.core.events.events.LowQualityExtractionEvent import com.correx.core.events.events.InferenceCompletedEvent import com.correx.core.events.events.InitialIntentEvent +import com.correx.core.events.events.SessionNamedEvent import com.correx.core.events.events.InferenceFailedEvent import com.correx.core.events.events.InferenceStartedEvent import com.correx.core.events.events.InferenceTimeoutEvent @@ -63,6 +64,7 @@ import com.correx.core.events.events.RepoMapComputedEvent import com.correx.core.events.events.RetryAttemptedEvent import com.correx.core.events.events.RetrySalvageDecidedEvent import com.correx.core.events.events.FailureTicketOpenedEvent +import com.correx.core.events.events.OutsidePathAccessGrantedEvent import com.correx.core.events.events.WorkspaceStateObservedEvent import com.correx.core.events.events.RiskAssessedEvent import com.correx.core.events.events.SourceFetchedEvent @@ -123,6 +125,7 @@ val eventModule = SerializersModule { subclass(PreemptRedirectEvent::class) subclass(PreemptRedirectBlockedEvent::class) subclass(InitialIntentEvent::class) + subclass(SessionNamedEvent::class) subclass(StageFailedEvent::class) subclass(StageCompletedEvent::class) subclass(TransitionExecutedEvent::class) @@ -151,6 +154,7 @@ val eventModule = SerializersModule { subclass(RetryAttemptedEvent::class) subclass(RetrySalvageDecidedEvent::class) subclass(FailureTicketOpenedEvent::class) + subclass(OutsidePathAccessGrantedEvent::class) subclass(RefinementIterationEvent::class) subclass(RepoMapComputedEvent::class) subclass(WorkspaceStateObservedEvent::class) diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/execution/StageOutcome.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/execution/StageOutcome.kt deleted file mode 100644 index 08bc1b21..00000000 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/execution/StageOutcome.kt +++ /dev/null @@ -1,18 +0,0 @@ -package com.correx.core.kernel.execution - -import com.correx.core.approvals.model.DomainApprovalRequest -import com.correx.core.artifacts.model.Artifact -import com.correx.core.validation.model.ValidationReport - -sealed interface StageOutcome { - data class Success(val artifact: Artifact) : StageOutcome - /** - * @param retryable set by the validator based on failure type; the orchestrator must read this - * and never override it. - * Structural errors (e.g. invalid graph topology) are not retryable. Transient errors may be. - */ - data class ValidationFailure(val report: ValidationReport, val retryable: Boolean) : StageOutcome - data class InferenceFailure(val reason: String, val retryable: Boolean) : StageOutcome - data class ApprovalRequired(val request: DomainApprovalRequest) : StageOutcome - data object Cancelled : StageOutcome -} diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ContextFeedback.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ContextFeedback.kt index 64ce6b77..83684a0b 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ContextFeedback.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ContextFeedback.kt @@ -9,6 +9,7 @@ import com.correx.core.context.model.ContextEntry import com.correx.core.context.model.ContextLayer import com.correx.core.context.model.EntryRole import com.correx.core.events.events.FailureTicketOpenedEvent +import com.correx.core.events.events.InitialIntentEvent import com.correx.core.events.events.RefinementIterationEvent import com.correx.core.events.events.RetryAttemptedEvent import com.correx.core.events.events.StoredEvent @@ -50,11 +51,31 @@ fun buildRecoveryTicketEntry(events: List, stageId: StageId): Conte val ticket = events .mapNotNull { it.payload as? FailureTicketOpenedEvent } .lastOrNull { it.routeTo == stageId } ?: return null - val content = "## Recovery ticket\n" + - "Stage '${ticket.stageId.value}' failed the '${ticket.gate}' gate (${ticket.category}) and " + - "lacks the '${ticket.requiredCapability}' capability to fix it, so the failure was routed to " + - "you. Repair the underlying cause, then control returns to verification.\n" + - "### Gate output\n${ticket.evidence}" + val content = if (ticket.escalated) { + // Tier 2: the owner loop couldn't settle it — a cross-file contract dispute. The arbiter holds + // the intent and reconciles ALL sides in one pass. + val intent = events.mapNotNull { it.payload as? InitialIntentEvent }.lastOrNull()?.intent + "## Contract arbitration ticket\n" + + "Stage '${ticket.stageId.value}' keeps failing the '${ticket.gate}' gate even after the " + + "file owners repaired their own layers — so this is NOT a bug in one file. The files named " + + "in the gate output below disagree on a shared contract (a prop/type/API shape): each side " + + "is internally consistent but they do not agree with each other, so fixing one re-breaks " + + "the other. You are the arbiter. Read EVERY file named below, decide the single authoritative " + + "shape" + (intent?.let { " that serves the intent" } ?: "") + ", then edit ALL of them in " + + "ONE pass so they agree. Do not rewrite files wholesale — change only what the contract " + + "requires. Then control returns to verification." + + (intent?.let { "\n### Intent (authoritative)\n$it" } ?: "") + + "\n### Gate output\n${ticket.evidence}" + } else { + "## Recovery ticket\n" + + "Stage '${ticket.stageId.value}' failed the '${ticket.gate}' gate (${ticket.category}) and " + + "lacks the '${ticket.requiredCapability}' capability to fix it, so the failure was routed to " + + "you — you own the file(s) the gate names below. Make the SMALLEST change that clears the " + + "specific error: read the current file, then patch only the offending lines (file_edit). Do " + + "NOT rewrite the file from scratch — a full regeneration discards your prior working code and " + + "usually reintroduces the same defect. Then control returns to verification.\n" + + "### Gate output\n${ticket.evidence}" + } return ContextEntry( id = ContextEntryId(UUID.randomUUID().toString()), layer = ContextLayer.L1, @@ -66,6 +87,47 @@ fun buildRecoveryTicketEntry(events: List, stageId: StageId): Conte ) } +/** + * The remaining-delta checklist (stage-termination design 2026-07-11 §1a). A forward-looking, + * shrinking list of the stage's *unmet* file-contract assertions — "what is left to make true", + * not "what I have done". Computed by the kernel as contract-minus-current-filesystem-reality and + * refreshed within the tool loop each time a write lands on a contract file, so the model watches + * items disappear as it satisfies them; an empty list is the completion signal the stage otherwise + * lacks. Reading files never changes it, which is what starves the re-read spiral. + * + * Pinned as a neverDrop entry by [com.correx.core.context.builder.DefaultContextPackBuilder] so the + * budget/dedup passes can never evict it — its whole value is surviving the truncation that gave the + * model amnesia in the first place. Returns null when nothing is failing (list empty ⇒ done). + * + * [items] are (target, assertionId, evidence) triples of the currently-failing assertions. + */ +fun buildRemainingDeltaEntry(items: List>): ContextEntry? { + if (items.isEmpty()) return null + val content = buildString { + append("## Remaining to finish this stage\n") + append( + "Make each item below true, then call stage_complete. Items disappear from this list " + + "as you satisfy them; when it is empty the stage is done. This list is recomputed " + + "from the filesystem — re-reading files does NOT change it, only writing/editing the " + + "named files does. Fix exactly these, do not re-explore.\n", + ) + items.forEach { (target, assertionId, evidence) -> + append("- [ ] ").append(target).append(" — ").append(assertionId) + if (evidence.isNotBlank()) append(": ").append(evidence) + append("\n") + } + }.trimEnd() + return ContextEntry( + id = ContextEntryId(UUID.randomUUID().toString()), + layer = ContextLayer.L1, + content = content, + sourceType = "remainingDelta", + sourceId = "remaining-delta", + tokenEstimate = content.length / 4, + role = EntryRole.SYSTEM, + ) +} + fun criticArtifactIds( events: List, graph: WorkflowGraph, diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultOrchestrationReducer.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultOrchestrationReducer.kt index d195ab18..e1aa403f 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultOrchestrationReducer.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultOrchestrationReducer.kt @@ -16,6 +16,10 @@ import com.correx.core.events.events.WorkflowStartedEvent import com.correx.core.events.orchestration.OrchestrationState import com.correx.core.events.orchestration.OrchestrationStatus +// Budget-key suffix distinguishing the tier-2 escalation (arbiter) budget from the tier-1 owner +// budget for the same failing stage. Shared with DefaultSessionOrchestrator (same package). +internal const val INTENT_BUDGET_SUFFIX = "#intent" + class DefaultOrchestrationReducer : OrchestrationReducer { override fun reduce( state: OrchestrationState, @@ -97,11 +101,16 @@ class DefaultOrchestrationReducer : OrchestrationReducer { // Charge the failing stage's recovery-route budget. routeAttempt is the no-progress count the // orchestrator computed (progress-aware), so fold it directly (idempotent under replay), and - // record the failure fingerprint the next route compares against. - is FailureTicketOpenedEvent -> state.copy( - recoveryRoutes = state.recoveryRoutes + (p.stageId.value to p.routeAttempt), - recoveryFailureFingerprints = state.recoveryFailureFingerprints + (p.stageId.value to p.fingerprint), - ) + // record the failure fingerprint the next route compares against. Keyed per TIER: the tier-1 + // owner loop and the tier-2 escalation (arbiter) hold independent budgets for the same failing + // stage, so an escalated ticket ("#intent" suffix) never spends the owner budget and vice versa. + is FailureTicketOpenedEvent -> { + val budgetKey = p.stageId.value + if (p.escalated) INTENT_BUDGET_SUFFIX else "" + state.copy( + recoveryRoutes = state.recoveryRoutes + (budgetKey to p.routeAttempt), + recoveryFailureFingerprints = state.recoveryFailureFingerprints + (budgetKey to p.fingerprint), + ) + } else -> state } diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultSessionOrchestrator.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultSessionOrchestrator.kt index b8226c19..b9cf320d 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultSessionOrchestrator.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/DefaultSessionOrchestrator.kt @@ -12,6 +12,8 @@ import com.correx.core.artifactstore.ArtifactStore import com.correx.core.events.events.ApprovalDecisionResolvedEvent import com.correx.core.events.events.ApprovalRequestedEvent import com.correx.core.events.events.ArtifactContentStoredEvent +import com.correx.core.events.events.FileWrittenEvent +import com.correx.core.events.events.ToolInvocationRequestedEvent import com.correx.core.events.events.ClarificationAnswer import com.correx.core.events.events.ClarificationAnsweredEvent import com.correx.core.events.events.ArtifactValidatedEvent @@ -63,6 +65,24 @@ private const val DEFAULT_MAX_REFINEMENT = 3 // per-gate retry budget — recovery is a purpose-built repair-then-reverify loop, not open-ended. private const val RECOVERY_ROUTE_BUDGET = 2 +// Tier-2 escalation budget: how many NO-PROGRESS routes the arbiter (recovery-stage-as-intent-holder) +// gets after the tier-1 owner loop exhausts, before the run fails terminally. Independent of the owner +// budget (keyed with INTENT_BUDGET_SUFFIX) so escalation is a genuine second chance, not a shared pool. +private const val INTENT_ROUTE_BUDGET = 2 + +// Transition ids for the failure-ticket loop. A ticket routes the failing gate's control to the +// repair owner (TICKET_ROUTE); the owner, once done, hands control straight back to the gate that +// opened the ticket (TICKET_RETURN). Both are kernel-driven, not plan edges — see ticketReturnMove. +internal val TICKET_ROUTE = TransitionId("ticket-route") +private val TICKET_RETURN = TransitionId("ticket-return") + +// Shortest evidence token treated as a file reference — guards against matching a written path's +// suffix on a trivially short token (an extension fragment, a one-char name). +private const val MIN_EVIDENCE_TOKEN = 5 + +// File-like tokens in gate evidence (tsc/build output): "src/App.tsx", "frontend/x.ts:12:5". +private val EVIDENCE_PATH_RE = Regex("""[\w./@-]+\.[A-Za-z0-9]+""") + // Deterministic gate id → capability the failing stage must hold to change the failure condition. // A gate not listed here is not eligible for recovery routing (retries in place as before). private val GATE_REQUIRED_CAPABILITY: Map = mapOf( @@ -161,12 +181,12 @@ class DefaultSessionOrchestrator( val resolved = resolveTransition( enriched.graph, enriched.sessionId, enriched.currentStageId, stageArtifacts, artifactContent, ) - // Recovery return-to-sender: a recovery stage is entered by kernel routing from whichever - // write-less stage failed a gate, so its exit must go BACK to that exact stage (re-run its - // gate) — not follow a static edge that would skip intervening stages. Derived from the latest - // failure ticket routed here (replay-deterministic). Overrides the resolved edge; an operator - // preempt below still wins. - val baseDecision = recoveryReturnMove(enriched) ?: resolved + // Ticket return-to-sender: a repair stage (the file's author, or the fallback recovery stage) + // is entered by kernel routing from whichever stage failed a gate, so its exit must go BACK to + // that exact gate (re-run it) — not follow a plan edge that would walk a regenerating + // implementer or skip intervening stages. Derived from the latest failure ticket routed here + // (replay-deterministic). Overrides the resolved edge; an operator preempt below still wins. + val baseDecision = ticketReturnMove(enriched) ?: resolved // Preemptive redirect override (freestyle graph re-routing): if the operator confirmed a // jump (a recorded, unconsumed PreemptRedirectEvent), it overrides the resolver's edge — // unless the target is unknown or its needs aren't satisfied, in which case it is blocked @@ -282,13 +302,24 @@ class DefaultSessionOrchestrator( } /** - * Open a [FailureTicketOpenedEvent] and route [stageId]'s failure to the graph's recovery stage, - * bounded by [RECOVERY_ROUTE_BUDGET]. Shared by the deterministic agency guard - * ([maybeRouteToRecovery]) and the review-gate salvage judge ([decideGateExhaustion] on - * [SalvageDecision.RECOVER]) so both routing paths reach the same recovery destination. + * Open a [FailureTicketOpenedEvent] and route [stageId]'s failure down a two-tier repair ladder. * - * Returns null when the graph declares no recovery stage (caller falls back to its legacy path); - * a terminal [StepResult] when the route budget is spent; otherwise the recovery stage's result. + * **Tier 1 (owner):** when the gate's evidence names a file a stage actually wrote, the ticket goes + * to that author ([resolveTicketOwner]) — it owns the code and the contract it declared, so it + * patches its own layer. Bounded by [RECOVERY_ROUTE_BUDGET]. + * + * **Tier 2 (arbiter / intent-holder):** when the owner loop spends its budget without settling the + * failure, the failure is a cross-file CONTRACT dispute no single owner can resolve (each side is + * internally consistent; they disagree with each other). The ticket escalates to the synthesized + * recovery stage ([findRecoveryStage]) recast as the intent-holder — given the initial intent as + * authority (see buildRecoveryTicketEntry) and told to reconcile ALL the named files in one pass. + * Bounded by its own independent [INTENT_ROUTE_BUDGET]. An orphan failure (no author) starts here. + * + * Both tiers charge progress-awarely: a route whose failure fingerprint changed from the previous + * one moved the needle, so it is FREE; only a route reproducing the same failure charges the tier's + * budget. Terminal only when every reachable tier is spent. Returns null when neither an owner nor a + * recovery stage exists (caller falls back to the legacy per-gate retry path). Shared by the agency + * guard ([maybeRouteToRecovery]) and the review-gate salvage judge ([decideGateExhaustion]). */ private suspend fun routeToRecovery( ctx: EnrichedExecutionContext, @@ -298,29 +329,38 @@ class DefaultSessionOrchestrator( reason: String, state: OrchestrationState, ): StepResult? { - val recoveryId = findRecoveryStage(ctx.graph, stageId) ?: return null // no recovery stage: legacy path - // Progress-aware charging (mirrors the per-gate retry budget, design §5): compare this - // failure's fingerprint to the previous route's. A changed fingerprint means the last - // recovery round moved the needle — fixed one cause and surfaced a different one — so the - // route is FREE (budget unchanged) and only the recorded fingerprint advances. Same - // fingerprint = the round changed nothing, so it charges the small route budget. Terminal - // only once the no-progress count is spent; the recovery→origin back-edge refinement guard - // (executeMove, cap = origin stage maxRetries) remains the ultimate loop bound for the - // pathological all-progress case. + val events = repositories.eventStore.read(ctx.sessionId) + val owner = resolveTicketOwner(events, reason) + ?.takeIf { it != stageId && requiredCapability in (ctx.graph.stages[it]?.allowedTools ?: emptySet()) } + val arbiter = findRecoveryStage(ctx.graph, stageId) val fingerprint = FailureFingerprint.of(reason) - val prev = state.recoveryFailureFingerprints[stageId.value] - val progressed = prev != null && prev != fingerprint - val used = state.recoveryRoutes[stageId.value] ?: 0 - if (!progressed && used >= RECOVERY_ROUTE_BUDGET) { + val ownerKey = stageId.value + val intentKey = stageId.value + INTENT_BUDGET_SUFFIX + + // (target, budgetKey, budget, escalated). Prefer the owner tier while it has budget; escalate to + // the arbiter tier once the owner loop is spent; null = no tier available now. + val route = when { + owner != null && !budgetExhausted(state, ownerKey, fingerprint, RECOVERY_ROUTE_BUDGET) -> + RouteTier(owner, ownerKey, RECOVERY_ROUTE_BUDGET, escalated = false) + arbiter != null && !budgetExhausted(state, intentKey, fingerprint, INTENT_ROUTE_BUDGET) -> + RouteTier(arbiter, intentKey, INTENT_ROUTE_BUDGET, escalated = true) + else -> null + } + if (route == null) { + if (owner == null && arbiter == null) return null // nothing to route to: legacy retry path return StepResult.Terminal( failWorkflow( ctx.sessionId, stageId, - "recovery route budget exhausted for stage ${stageId.value} (gate=$gate): $reason", + "repair ladder exhausted for stage ${stageId.value} (gate=$gate): $reason", retryExhausted = true, ), ) } + + val prev = state.recoveryFailureFingerprints[route.budgetKey] + val progressed = prev != null && prev != fingerprint + val used = state.recoveryRoutes[route.budgetKey] ?: 0 val routeAttempt = if (progressed) used else used + 1 emit( ctx.sessionId, @@ -330,24 +370,51 @@ class DefaultSessionOrchestrator( gate = gate, category = ticketCategory(gate), requiredCapability = requiredCapability, - routeTo = recoveryId, + routeTo = route.target, evidence = reason, routeAttempt = routeAttempt, fingerprint = fingerprint, + escalated = route.escalated, ), ) log.info( - "[Orchestrator] failure ticket session={} stage={} gate={} lacks={} -> recovery={} " + + "[Orchestrator] failure ticket session={} stage={} gate={} lacks={} -> {}={} " + "(charged={}/{}, progressed={})", ctx.sessionId.value, stageId.value, gate, requiredCapability, - recoveryId.value, routeAttempt, RECOVERY_ROUTE_BUDGET, progressed, + if (route.escalated) "arbiter" else "owner", route.target.value, + routeAttempt, route.budget, progressed, ) val advancedTo = advanceStage( ctx.sessionId, stageId, - TransitionDecision.Move(transitionId = TransitionId("recovery-route"), to = recoveryId), + TransitionDecision.Move(transitionId = TICKET_ROUTE, to = route.target), ) - return enterStage(ctx.copy(currentStageId = advancedTo), recoveryId) + return enterStage(ctx.copy(currentStageId = advancedTo), route.target) + } + + /** A chosen rung of the repair ladder: where to route, which budget it charges, and its tier. */ + private data class RouteTier( + val target: StageId, + val budgetKey: String, + val budget: Int, + val escalated: Boolean, + ) + + /** + * Progress-aware budget check for one ladder tier. A route whose failure fingerprint changed from + * the previous route on the same [key] made progress (fixed one cause, surfaced another) → NOT + * exhausted (progress is free). Only a no-progress streak that reaches [budget] is exhausted. + */ + private fun budgetExhausted( + state: OrchestrationState, + key: String, + fingerprint: String, + budget: Int, + ): Boolean { + val prev = state.recoveryFailureFingerprints[key] + val progressed = prev != null && prev != fingerprint + val used = state.recoveryRoutes[key] ?: 0 + return !progressed && used >= budget } /** @@ -361,23 +428,49 @@ class DefaultSessionOrchestrator( }?.key /** - * When the current stage is a recovery stage that has finished, route it back to the stage the - * failure came from — the origin of the latest [FailureTicketOpenedEvent] routed here — so the - * gate that triggered the route is re-run. This makes recovery correct for arbitrary graph - * topology (a write-less gate anywhere routes here and returns to exactly its own stage) rather - * than relying on a static recovery→terminal edge that would skip intervening stages. Returns - * null when the current stage is not a recovery stage or no ticket routed here (then the - * resolver's normal edge — the compiler's fallback recovery→terminal — is used). + * Route-to-owner resolution: map the failing gate's [evidence] (build/tsc output that names + * files, e.g. "src/App.tsx: TS2322") to the stage that most recently WROTE one of those files, + * via the recorded manifest chain (ToolInvocationRequestedEvent.stageId ← FileWrittenEvent by + * invocationId → path). The last write of a named file wins — its author is who to hand the + * ticket to. Pure over recorded events (replay-deterministic). Null when the evidence names no + * written file (an orphan failure — caller falls back to the synthesized recovery stage). */ - private fun recoveryReturnMove(ctx: EnrichedExecutionContext): TransitionDecision.Move? { - val cfg = ctx.graph.stages[ctx.currentStageId] ?: return null - val isRecovery = cfg.metadata["role"] == "recovery" || ctx.currentStageId.value == "recovery" - if (!isRecovery) return null - val origin = repositories.eventStore.read(ctx.sessionId) - .mapNotNull { it.payload as? FailureTicketOpenedEvent } + private fun resolveTicketOwner(events: List, evidence: String): StageId? { + val tokens = EVIDENCE_PATH_RE.findAll(evidence) + .map { it.value.substringBefore(':').replace('\\', '/') } + .filter { it.length >= MIN_EVIDENCE_TOKEN } + .toSet() + if (tokens.isEmpty()) return null + val invToStage = events.mapNotNull { it.payload as? ToolInvocationRequestedEvent } + .associate { it.invocationId to it.stageId } + return events.mapNotNull { it.payload as? FileWrittenEvent } + .filter { it.postImageHash != null } + .lastOrNull { fw -> + val norm = fw.path.replace('\\', '/') + invToStage.containsKey(fw.invocationId) && tokens.any { norm == it || norm.endsWith("/$it") } + } + ?.let { invToStage[it.invocationId] } + } + + /** + * Return-to-sender for a routed repair stage. When the current stage was entered by a failure + * ticket ([TICKET_ROUTE]) rather than a plan edge, its exit goes BACK to the gate that opened the + * ticket (origin of the latest [FailureTicketOpenedEvent] routed here) so that gate re-runs. This + * holds for any repair destination — the file's author OR the fallback recovery stage — and for + * arbitrary topology, instead of following the plan's forward edge (which would walk a + * regenerating implementer, or skip intervening stages). Returns null when the stage was NOT + * entered via a ticket (a normal forward visit — the resolver's edge is used). + */ + private fun ticketReturnMove(ctx: EnrichedExecutionContext): TransitionDecision.Move? { + val events = repositories.eventStore.read(ctx.sessionId) + val enteredViaTicket = events.mapNotNull { it.payload as? TransitionExecutedEvent } + .lastOrNull { it.to == ctx.currentStageId } + ?.transitionId == TICKET_ROUTE + if (!enteredViaTicket) return null + val origin = events.mapNotNull { it.payload as? FailureTicketOpenedEvent } .lastOrNull { it.routeTo == ctx.currentStageId } ?.stageId ?: return null - return TransitionDecision.Move(transitionId = TransitionId("recovery-return"), to = origin) + return TransitionDecision.Move(transitionId = TICKET_RETURN, to = origin) } /** Returns the cached validated artifact content for the given session + artifact id, or null if absent. */ diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ProcessStaticAnalysisRunner.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ProcessStaticAnalysisRunner.kt index 0fa225a6..4c86727d 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ProcessStaticAnalysisRunner.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/ProcessStaticAnalysisRunner.kt @@ -1,5 +1,6 @@ package com.correx.core.kernel.orchestration +import com.correx.core.tools.process.ChildProcess import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.TimeoutCancellationException import kotlinx.coroutines.async @@ -26,7 +27,7 @@ class ProcessStaticAnalysisRunner( val argv = command.trim().split(WHITESPACE).filter { it.isNotEmpty() } if (argv.isEmpty()) return@withContext StaticAnalysisRunResult(EXIT_NOT_RUN, "empty command") val process = runCatching { - ProcessBuilder(argv).directory(workingDir.toFile()).redirectErrorStream(true).start() + ChildProcess.builder(argv, workingDir.toFile()).redirectErrorStream(true).start() }.getOrElse { return@withContext StaticAnalysisRunResult(EXIT_NOT_RUN, "failed to start '$command': ${it.message}") } diff --git a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt index da7727af..b5d516a2 100644 --- a/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt +++ b/core/kernel/src/main/kotlin/com/correx/core/kernel/orchestration/SessionOrchestrator.kt @@ -71,6 +71,7 @@ import com.correx.core.events.events.InferenceTimeoutEvent import com.correx.core.events.events.NewEvent import com.correx.core.events.events.OrchestrationPausedEvent import com.correx.core.events.events.OrchestrationResumedEvent +import com.correx.core.events.events.OutsidePathAccessGrantedEvent import com.correx.core.events.events.RiskAssessedEvent import com.correx.core.events.events.ExecutionPlanLockedEvent import com.correx.core.events.events.StageCheckpointFailedEvent @@ -167,13 +168,21 @@ import com.correx.core.journal.DefaultDecisionJournalRepository import com.correx.core.kernel.orchestration.subagent.SubagentRunner import org.slf4j.LoggerFactory import java.nio.file.Files +import java.nio.file.Path import java.nio.file.Paths import java.util.* import java.util.concurrent.* import java.util.concurrent.atomic.* import kotlin.coroutines.cancellation.CancellationException -private const val MAX_TOOL_ROUNDS = 10 +// Ceiling on agentic tool rounds per stage entry. Originally 10 as a tight leash on early local +// models that would thrash tools; the model now drives tools reliably, and the tight cap started +// starving legitimate multi-step work — notably the recovery stage, which spends most rounds +// re-investigating (list_dir/read/rebuild) before it reaches the write, and got bounced out before +// acting on a correct diagnosis. The real runaway guards are the read-loop and rejection-loop +// nudges (both fire after 3 unproductive rounds), so this ceiling only needs to be generous enough +// that productive stages aren't cut off mid-work. +private const val MAX_TOOL_ROUNDS = 30 // Consecutive read-only tool rounds (no file_write/file_edit) that still owe a file_written // artifact before we force the write nudge. A model that keeps calling read tools every round @@ -251,6 +260,8 @@ private const val REVIEW_BLOCK_RETRY_CAP = 20 private const val REVIEW_OBJECTIVE_CAP = 4_000 private const val STATIC_ANALYSIS_SUMMARY_CAP = 2_000 private const val STATIC_ANALYSIS_FEEDBACK_CAP = 6_000 +/** Profile command alias run before a build gate to make it runnable (e.g. `npm ci`). */ +private const val SETUP_COMMAND_ALIAS = "setup" // KindInference kinds for build manifests — a written file of one of these declares a buildable // project toolchain even before any source lands, so the auto build gate fires (see @@ -414,7 +425,15 @@ abstract class SessionOrchestrator( ), ) } ?: emptyList() - val promptEntries = stageConfig.metadata["promptInline"] + // A stage re-entered to repair a gate failure has its own (often generative/"scaffold") prompt + // SUPPRESSED: that mandate is what drove it to overwrite real files with stubs on re-entry. The + // recovery-ticket entry (buildRecoveryTicketEntry) is the sole mandate here — it already carries + // "make the smallest change, don't rewrite from scratch" plus the gate evidence. See + // mandateSuppressedByTicket. + val promptEntries = if (mandateSuppressedByTicket(sessionId, stageId, stageConfig)) { + emptyList() + } else { + stageConfig.metadata["promptInline"] ?.takeIf { it.isNotBlank() } ?.let { text -> listOf( @@ -457,6 +476,7 @@ abstract class SessionOrchestrator( ) } ?: emptyList() + } val llmEmittedSlots = stageConfig.produces.filter { it.kind.llmEmitted } require(llmEmittedSlots.size <= 1) { @@ -553,11 +573,19 @@ abstract class SessionOrchestrator( } else { emptyList() } + // Remaining-delta checklist (stage-termination design 2026-07-11 §1a): the stage's currently + // unmet file-contract assertions as a forward-looking, shrinking TODO. Injected pinned + // (neverDrop) so it survives the budget/dedup passes that otherwise wipe the model's memory of + // progress; refreshed inside the tool loop each time a write lands (§1b, cache-until-write). + var remainingDeltaResults = evaluateStageContract(sessionId, stageId, stageConfig, effectives) + val remainingDeltaEntries = remainingDeltaResults + ?.let { buildRemainingDeltaEntry(contractFailureItems(it)) } + ?.let { listOf(it) } ?: emptyList() var accumulatedEntries = stampBuckets( systemPrompt + profileEntries + projectProfileEntries + agentInstructionsEntries + journalEntries + repoMapEntries + claimedTaskEntries + needsEntries + schemaEntries + vocabularyEntries + promptEntries + steeringEntries + - clarificationEntries + retryFeedbackEntries + recoveryTicketEntries, + clarificationEntries + retryFeedbackEntries + recoveryTicketEntries + remainingDeltaEntries, ) val contextPack = runCatching { contextPackBuilder.build( @@ -767,6 +795,22 @@ abstract class SessionOrchestrator( } else { consecutiveRejectedRounds = 0 } + // Refresh the remaining-delta checklist only when a write actually landed this round + // (cache-until-write, §1b): reads can never change the contract verdict, so re-evaluating + // on them would just burn filesystem work and re-emit identical events. On a successful + // write, recompute contract-minus-reality and swap the pinned entry so the model watches + // the list shrink toward the empty=done signal it previously lacked. + val wroteThisRound = inferenceResult.response.toolCalls.any { it.function.name in WRITE_TOOL_NAMES } && + toolEntries.any { + it.role == EntryRole.TOOL && + !it.content.startsWith("ERROR:") && !it.content.startsWith("BLOCKED:") + } + if (wroteThisRound) { + remainingDeltaResults = evaluateStageContract(sessionId, stageId, stageConfig, effectives) + val refreshed = remainingDeltaResults?.let { buildRemainingDeltaEntry(contractFailureItems(it)) } + accumulatedEntries = accumulatedEntries.filterNot { it.sourceType == "remainingDelta" } + + listOfNotNull(refreshed) + } currentContext = contextPackBuilder.build( id = ContextPackId(UUID.randomUUID().toString()), sessionId = sessionId, @@ -1084,6 +1128,19 @@ abstract class SessionOrchestrator( ) val tool = effectives.registry?.resolve(toolCall.function.name) val tier = tool?.tier ?: Tier.T2 + // Out-of-workspace read access: a read whose target lexically resolves outside the + // workspace root. The intent plane raises PROMPT_USER for it; once the operator approves + // (recorded as OutsidePathAccessGrantedEvent), the same path this session skips the prompt + // and the tool jail is widened for it. Writes never take this path (isRead gate). + val isRead = tool?.requiredCapabilities?.contains(ToolCapability.FILE_READ) == true && + tool.requiredCapabilities.contains(ToolCapability.FILE_WRITE) != true + val outsideReadTarget = if (isRead) { + outsideWorkspaceTarget(parameters, effectives.policy?.workspaceRoot) + } else { + null + } + val grantedOutside = grantedOutsidePaths(sessionId) + val alreadyGranted = outsideReadTarget != null && outsideReadTarget.toString() in grantedOutside emit( sessionId, ToolInvocationRequestedEvent( @@ -1202,8 +1259,9 @@ abstract class SessionOrchestrator( // A steering note attached to a human approval is captured here and injected after the // tool result, so the same-stage loop re-infers with it and the model acts on the note. var approvalNote: String? = null - if (tier.isAtMost(Tier.T1) && !plane2Prompts) { - // no approval needed + if ((tier.isAtMost(Tier.T1) && !plane2Prompts) || alreadyGranted) { + // no approval needed — either within the auto-approve tier, or this out-of-workspace + // read path was already approved earlier this session (this-path-this-session). } else { // Grants in effect = this session's own (SESSION/STAGE) unioned with the // cross-session ledger (PROJECT/GLOBAL). projectId is derived from the bound @@ -1329,7 +1387,24 @@ abstract class SessionOrchestrator( emit(sessionId, OrchestrationResumedEvent(sessionId, stageId)) } } - val result = executor.execute(request) + // Reaching here means the call is authorized (all deny/block/pending paths returned early). + // For a first-time out-of-workspace read, record the grant so future reads of the same path + // this session skip the prompt (invariants #8/#9 — replay reads the recorded fact). Then + // widen the executed request's jail with every path approved outside the workspace. + if (outsideReadTarget != null && !alreadyGranted) { + emit( + sessionId, + OutsidePathAccessGrantedEvent( + sessionId = sessionId, + stageId = stageId, + path = outsideReadTarget.toString(), + ), + ) + } + val executedRequest = outsideReadTarget + ?.let { request.copy(grantedPaths = grantedOutside + it.toString()) } + ?: request + val result = executor.execute(executedRequest) // Store ProcessResult artifact for every shell execution outcome for (slot in processResultSlots) { @@ -1496,6 +1571,33 @@ abstract class SessionOrchestrator( .fold(projection.initial()) { acc, event -> projection.apply(acc, event) } } + /** + * The out-of-workspace path a read tool targets, or null if the read stays inside the workspace + * (or there's no workspace policy / path arg). Resolution is purely lexical (absolute-normalize, + * or resolve-against-root for relatives) so the containment decision is deterministic and never + * touches the filesystem. Symlink escapes and privileged locations are caught upstream by the + * plane-2 PathContainmentRule (BLOCK before any prompt), and by the tool's own realize() jail. + */ + private fun outsideWorkspaceTarget(parameters: Map, workspaceRoot: Path?): Path? { + workspaceRoot ?: return null + val raw = parameters["path"] as? String ?: return null + val candidate = runCatching { Path.of(raw) }.getOrNull() ?: return null + val root = workspaceRoot.toAbsolutePath().normalize() + val resolved = (if (candidate.isAbsolute) candidate else root.resolve(candidate)).normalize() + return resolved.takeUnless { it.startsWith(root) } + } + + /** + * Paths outside the workspace the operator has approved this session (folded from + * [OutsidePathAccessGrantedEvent]). Replay-safe: reads the log, never re-prompts or re-probes. + */ + private fun grantedOutsidePaths(sessionId: SessionId): Set = + eventStore.read(sessionId) + .mapNotNull { it.payload as? OutsidePathAccessGrantedEvent } + .filter { it.sessionId == sessionId } + .map { it.path } + .toSet() + /** * Returns true when the agent must be offered only read-only tools on the next inference turn. * Active from the moment a READ_BEFORE_WRITE block lands until a FILE_READ completion follows it. @@ -1721,6 +1823,25 @@ abstract class SessionOrchestrator( ) } + /** + * True when [stageId]'s own prompt must be suppressed because the stage was entered to repair a + * gate failure. That happens when the most recent transition INTO this stage was a ticket route + * ([TICKET_ROUTE], same entry check as DefaultSessionOrchestrator.ticketReturnMove) — the failure + * ticket routes an owner stage back to fix a file it authored, and its original generative mandate + * ("scaffold the project") is exactly what drives it to overwrite that file with a stub. The + * recovery-ticket entry ([buildRecoveryTicketEntry]) becomes the sole mandate instead. + * + * The purpose-built recovery arbiter (role=recovery) is excluded: it is also ticket-entered but + * its own prompt IS a tuned repair prompt, so it keeps it. Replay-safe (recorded events only). + */ + private fun mandateSuppressedByTicket(sessionId: SessionId, stageId: StageId, stageConfig: StageConfig): Boolean { + if (stageConfig.metadata["role"] == "recovery") return false + return eventStore.read(sessionId) + .mapNotNull { it.payload as? TransitionExecutedEvent } + .lastOrNull { it.to == stageId } + ?.transitionId == TICKET_ROUTE + } + /** * Best real path in the repo map that shares [guessed]'s filename, ranked by shared trailing * path segments — so a wrong package prefix (e.g. `com/correx/server/...` for the real @@ -2304,27 +2425,35 @@ abstract class SessionOrchestrator( * empty/missing/malformed file rather than read-looping. COMPILER assertions are recorded as * skipped here; the static-analysis command gate is their authoritative check. */ - private suspend fun runContractGate( + /** + * Evaluate the stage's file contract against current filesystem reality and record the verdict + * in a [ContractGateEvaluatedEvent] (invariant #9). Returns the per-assertion results, or null + * when the gate does not apply (no evaluator, no workspace root, no file_written produces, or no + * paths). Shared by the post-stage [runContractGate] and the within-loop remaining-delta refresh + * ([buildRemainingDeltaEntry]) so both read the same contract-minus-reality computation from a + * single evaluation + single event emission. + */ + private suspend fun evaluateStageContract( sessionId: SessionId, stageId: StageId, stageConfig: StageConfig, effectives: RunEffectives, - ): StageExecutionResult { - val evaluator = contractAssertionEvaluator ?: return StageExecutionResult.Success(emptyList()) - val workspaceRoot = effectives.policy?.workspaceRoot ?: return StageExecutionResult.Success(emptyList()) - if (stageConfig.produces.none { it.kind.id == "file_written" }) return StageExecutionResult.Success(emptyList()) + ): List? { + val evaluator = contractAssertionEvaluator ?: return null + val workspaceRoot = effectives.policy?.workspaceRoot ?: return null + if (stageConfig.produces.none { it.kind.id == "file_written" }) return null // Option A ∪ B: the files the stage actually wrote (runtime manifest) unioned with the // concrete files the plan declared it would produce (expectedFiles). A declared file that // was never written survives here and fails file_exists — that is the missing-file catch. val paths = (stageWrittenPaths(sessionId, stageId) + stageConfig.expectedFiles).distinct() - if (paths.isEmpty()) return StageExecutionResult.Success(emptyList()) + if (paths.isEmpty()) return null val assertions = paths.flatMap { path -> KindContractTable.assertionsFor(KindInference.kindFor(path) ?: "", path) } val results = assertions.map { a -> - val v = contractAssertionEvaluator.evaluate(workspaceRoot, a) + val v = evaluator.evaluate(workspaceRoot, a) ContractAssertionResult( assertionId = a.id, target = a.target, @@ -2335,6 +2464,21 @@ abstract class SessionOrchestrator( ) } emit(sessionId, ContractGateEvaluatedEvent(sessionId, stageId, results)) + return results + } + + /** The currently-failing assertions as (target, assertionId, evidence) triples for the checklist. */ + private fun contractFailureItems(results: List): List> = + results.filterNot { it.passed }.map { Triple(it.target, it.assertionId, it.evidence) } + + private suspend fun runContractGate( + sessionId: SessionId, + stageId: StageId, + stageConfig: StageConfig, + effectives: RunEffectives, + ): StageExecutionResult { + val results = evaluateStageContract(sessionId, stageId, stageConfig, effectives) + ?: return StageExecutionResult.Success(emptyList()) val failures = results.filterNot { it.passed } if (failures.isEmpty()) return StageExecutionResult.Success(emptyList()) @@ -2528,6 +2672,8 @@ abstract class SessionOrchestrator( return StageExecutionResult.Success(emptyList()) } + runSetupCommand(sessionId, stageId, workspaceRoot, profileCommands, runner)?.let { return it } + val run = runner.run(workspaceRoot, command) val finding = StaticAnalysisFinding(command, run.exitCode, run.output.takeLast(STATIC_ANALYSIS_SUMMARY_CAP)) emit(sessionId, StaticAnalysisCompletedEvent(sessionId, stageId, listOf(finding))) @@ -2546,6 +2692,41 @@ abstract class SessionOrchestrator( ) } + /** + * Run the operator-declared `setup` command (profile alias [SETUP_COMMAND_ALIAS]) before a build + * gate, when one is configured. A build like `tsc && vite build` fails spuriously if the project + * was never prepared (deps uninstalled, codegen not run) — the funnel reads that as a code failure + * and routes into futile repair. The operator names the fix in their profile (`setup = "npm ci"`, + * `"cargo fetch"`, `"./gradlew dependencies"`, …); the kernel stays ecosystem-agnostic. Runs in the + * workspace root, recorded as a [StaticAnalysisCompletedEvent] (invariant #9). Returns a + * [StageExecutionResult.Failure] when setup itself fails (a real, actionable error), else null (no + * setup configured, or it succeeded — proceed to the build). + */ + private suspend fun runSetupCommand( + sessionId: SessionId, + stageId: StageId, + workspaceRoot: Path, + profileCommands: Map, + runner: StaticAnalysisRunner, + ): StageExecutionResult? { + val setup = profileCommands[SETUP_COMMAND_ALIAS]?.takeIf { it.isNotBlank() } ?: return null + log.info( + "[Orchestrator] running setup command before build gate session={} stage={} command='{}'", + sessionId.value, stageId.value, setup, + ) + val run = runner.run(workspaceRoot, setup) + val finding = StaticAnalysisFinding(setup, run.exitCode, run.output.takeLast(STATIC_ANALYSIS_SUMMARY_CAP)) + emit(sessionId, StaticAnalysisCompletedEvent(sessionId, stageId, listOf(finding))) + if (finding.clean) return null + return StageExecutionResult.Failure( + "stage ${stageId.value} setup command failed before its build gate. " + + "Fix these before proceeding:\n\n$ $setup (exit ${run.exitCode})\n" + + run.output.takeLast(STATIC_ANALYSIS_FEEDBACK_CAP), + retryable = true, + gate = "execution", + ) + } + /** * Gate 3 — semantic (LLM) review (staged-verification § Gate 3). For a stage that opts in * (`semanticReview`), the injected [SemanticReviewer] reads the files the stage actually wrote @@ -3441,8 +3622,17 @@ internal sealed interface InferenceResult { * becomes its content string; objects and arrays-of-objects keep their JSON text for tools that * re-parse them (e.g. task_decompose). Malformed JSON yields an empty map. */ +// Lenient fallback for small local models (gemma) that drift out of strict JSON on large tool +// args — most commonly bare/unquoted keys (`path:` instead of `"path":`) and unquoted string +// values. Strict parse is tried first; only genuinely-off syntax hits this. Triple-quoted +// Python-style values (`"""..."""`) still fail here — that ambiguity isn't worth a fragile +// hand-rolled repair; the loud "unparseable" path below tells the model to fix its format. +private val lenientJson = Json { isLenient = true } + internal fun parseToolArguments(arguments: String): Map = runCatching { - val obj = Json.parseToJsonElement(arguments) as? JsonObject ?: return@runCatching emptyMap() + val element = runCatching { Json.parseToJsonElement(arguments) } + .getOrElse { lenientJson.parseToJsonElement(arguments) } + val obj = element as? JsonObject ?: return@runCatching emptyMap() obj.entries.associate { (k, v) -> k to when { v is JsonPrimitive -> v.content diff --git a/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/ParseToolArgumentsTest.kt b/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/ParseToolArgumentsTest.kt index 612a96b4..5fcad838 100644 --- a/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/ParseToolArgumentsTest.kt +++ b/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/ParseToolArgumentsTest.kt @@ -39,6 +39,16 @@ class ParseToolArgumentsTest { assertTrue(parseToolArguments("[]").isEmpty()) // top-level non-object } + @Test + fun `lenient fallback recovers unquoted keys from gemma drift`() { + // Small local models drift out of strict JSON on large args, most commonly bare keys. + // Strict parse fails; the lenient fallback still yields both params so file_write no + // longer sees an empty map and lies "Missing 'path'". + val params = parseToolArguments("""{path: "styles/x.css", content: "body {}"}""") + assertEquals("styles/x.css", params["path"]) + assertEquals("body {}", params["content"]) + } + @Test fun `decompose preview renders the epic and children with dependency labels`() { // Args arrive flattened: nested tasks/parent are JSON text, as in a real call. diff --git a/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/RemainingDeltaEntryTest.kt b/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/RemainingDeltaEntryTest.kt new file mode 100644 index 00000000..51d27fb7 --- /dev/null +++ b/core/kernel/src/test/kotlin/com/correx/core/kernel/orchestration/RemainingDeltaEntryTest.kt @@ -0,0 +1,33 @@ +package com.correx.core.kernel.orchestration + +import com.correx.core.context.model.EntryRole +import org.junit.jupiter.api.Assertions.assertEquals +import org.junit.jupiter.api.Assertions.assertNull +import org.junit.jupiter.api.Assertions.assertTrue +import org.junit.jupiter.api.Test + +class RemainingDeltaEntryTest { + + @Test + fun `empty failing set yields no entry (list empty means done)`() { + assertNull(buildRemainingDeltaEntry(emptyList())) + } + + @Test + fun `entry lists each failing assertion as an unchecked forward item`() { + val entry = buildRemainingDeltaEntry( + listOf( + Triple("frontend/src/views/TaskView.tsx", "exports_default_component", "no default export"), + Triple("frontend/src/views/SessionView.tsx", "exports_default_component", "no default export"), + ), + )!! + assertEquals("remainingDelta", entry.sourceType) + assertEquals(EntryRole.SYSTEM, entry.role) + // Forward-looking framing, not a history of what was done. + assertTrue(entry.content.contains("Remaining to finish this stage")) + assertTrue(entry.content.contains("- [ ] frontend/src/views/TaskView.tsx — exports_default_component")) + assertTrue(entry.content.contains("SessionView.tsx")) + // Must state that re-reading doesn't change it — the spiral-starver. + assertTrue(entry.content.contains("re-reading files does NOT change it", ignoreCase = true)) + } +} diff --git a/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummary.kt b/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummary.kt index 7946bbc4..825ff103 100644 --- a/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummary.kt +++ b/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummary.kt @@ -7,6 +7,7 @@ import kotlinx.serialization.Serializable @Serializable data class SessionSummary( val sessionId: SessionId, + val name: String?, val intent: String?, val workflowId: String?, val status: String, diff --git a/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummaryProjector.kt b/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummaryProjector.kt index 8973b5a5..3ca2e528 100644 --- a/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummaryProjector.kt +++ b/core/sessions/src/main/kotlin/com/correx/core/sessions/SessionSummaryProjector.kt @@ -2,6 +2,7 @@ package com.correx.core.sessions import com.correx.core.events.events.InitialIntentEvent import com.correx.core.events.events.OrchestrationPausedEvent +import com.correx.core.events.events.SessionNamedEvent import com.correx.core.events.events.StoredEvent import com.correx.core.events.events.TransitionExecutedEvent import com.correx.core.events.events.WorkflowCompletedEvent @@ -12,6 +13,7 @@ import com.correx.core.events.types.SessionId class SessionSummaryProjector { fun project(sessionId: SessionId, events: List): SessionSummary { var intent: String? = null + var name: String? = null var workflowId: String? = null var stageCount = 0 var status = "UNKNOWN" @@ -21,6 +23,7 @@ class SessionSummaryProjector { events.forEach { event -> when (val p = event.payload) { is InitialIntentEvent -> intent = p.intent + is SessionNamedEvent -> name = p.name is WorkflowStartedEvent -> { workflowId = p.workflowId status = SessionStatus.ACTIVE.name @@ -35,6 +38,7 @@ class SessionSummaryProjector { return SessionSummary( sessionId = sessionId, + name = name, intent = intent, workflowId = workflowId, status = status, diff --git a/core/sessions/src/test/kotlin/com/correx/core/sessions/SessionSummaryProjectorTest.kt b/core/sessions/src/test/kotlin/com/correx/core/sessions/SessionSummaryProjectorTest.kt index c52e49c8..0534a30e 100644 --- a/core/sessions/src/test/kotlin/com/correx/core/sessions/SessionSummaryProjectorTest.kt +++ b/core/sessions/src/test/kotlin/com/correx/core/sessions/SessionSummaryProjectorTest.kt @@ -2,6 +2,7 @@ package com.correx.core.sessions import com.correx.core.events.events.EventMetadata import com.correx.core.events.events.InitialIntentEvent +import com.correx.core.events.events.SessionNamedEvent import com.correx.core.events.events.StoredEvent import com.correx.core.events.events.TransitionExecutedEvent import com.correx.core.events.types.EventId @@ -58,6 +59,15 @@ class SessionSummaryProjectorTest { assertEquals("add auth layer", summary.intent) } + @Test + fun `SessionNamedEvent surfaces the human name in the summary`() { + val event = stored(SessionNamedEvent(sessionId, "Add auth layer", timestampMs = 0L)) + + val summary = projector.project(sessionId, listOf(event)) + + assertEquals("Add auth layer", summary.name) + } + @Test fun `two TransitionExecutedEvents yield stageCount of two`() { val t1 = stored( diff --git a/core/talkie/src/main/kotlin/com/correx/core/talkie/TalkieFacade.kt b/core/talkie/src/main/kotlin/com/correx/core/talkie/TalkieFacade.kt index c3c2ba3f..16412705 100644 --- a/core/talkie/src/main/kotlin/com/correx/core/talkie/TalkieFacade.kt +++ b/core/talkie/src/main/kotlin/com/correx/core/talkie/TalkieFacade.kt @@ -1,5 +1,10 @@ package com.correx.core.talkie +import com.correx.core.context.model.CompressionMetadata +import com.correx.core.context.model.ContextEntry +import com.correx.core.context.model.ContextLayer +import com.correx.core.context.model.ContextPack +import com.correx.core.context.model.EntryRole import com.correx.core.events.events.ChatTurnEvent import com.correx.core.events.events.ChatTurnRole import com.correx.core.events.events.ContextTruncatedEvent @@ -10,9 +15,12 @@ import com.correx.core.events.events.L3RetrievedHit import com.correx.core.events.events.NewEvent import com.correx.core.events.events.ToolInvocationRequestedEvent import com.correx.core.events.events.TalkieNarrationEvent +import com.correx.core.events.events.SessionNamedEvent import com.correx.core.events.events.SteeringNoteAddedEvent import com.correx.core.events.events.WorkflowProposedEvent import com.correx.core.events.stores.EventStore +import com.correx.core.events.types.ContextEntryId +import com.correx.core.events.types.ContextPackId import com.correx.core.events.types.EventId import com.correx.core.events.types.InferenceRequestId import com.correx.core.events.types.SessionId @@ -42,6 +50,11 @@ private val log = LoggerFactory.getLogger(DefaultTalkieFacade::class.java) private const val MAX_NARRATED_ACTIONS = 6 private const val ACTION_ARG_MAX = 60 +// Session-naming caps. +private const val NAME_INTENT_MAX = 2000 +private const val NAME_MAX_CHARS = 64 +private const val TOKEN_CHARS = 4 + interface TalkieFacade { suspend fun onUserInput( sessionId: SessionId, @@ -50,6 +63,13 @@ interface TalkieFacade { ): TalkieResponse suspend fun narrate(sessionId: SessionId, trigger: NarrationTrigger) + + /** + * Derives a short, human-readable title for the session from its initial [intent] and records + * it once as a [SessionNamedEvent]. Nondeterministic (an LLM call) so the result is captured as + * an event, never re-derived on replay (invariant #9). No-op on blank intent or empty output. + */ + suspend fun nameSession(sessionId: SessionId, intent: String) } class DefaultTalkieFacade( @@ -254,6 +274,80 @@ class DefaultTalkieFacade( ) } + override suspend fun nameSession(sessionId: SessionId, intent: String) { + if (intent.isBlank()) return + + val instruction = buildString { + append("Produce a short, specific title (3-6 words, Title Case, no quotes, no trailing ") + append("punctuation) that names the task described below. Reply with the title only.\n\n") + append("Task: ") + append(intent.trim().take(NAME_INTENT_MAX)) + } + val contextPack = ContextPack( + id = ContextPackId("${sessionId.value}-naming-pack"), + sessionId = sessionId, + stageId = StageId.NONE, + layers = mapOf( + ContextLayer.L0 to listOf( + ContextEntry( + id = ContextEntryId(UUID.randomUUID().toString()), + layer = ContextLayer.L0, + content = instruction, + sourceType = "sessionNaming", + sourceId = sessionId.value, + tokenEstimate = (instruction.length / TOKEN_CHARS).coerceAtLeast(1), + role = EntryRole.USER, + ), + ), + ), + budgetUsed = (instruction.length / TOKEN_CHARS).coerceAtLeast(1), + budgetLimit = config.tokenBudget.limit, + compressionMetadata = CompressionMetadata(appliedStrategies = listOf("SessionNaming")), + ) + + val provider = inferenceRouter.route(StageId.NONE, setOf(ModelCapability.General), config.narrationModelId) + val inferenceRequest = InferenceRequest( + requestId = InferenceRequestId(UUID.randomUUID().toString()), + sessionId = sessionId, + stageId = StageId.NONE, + contextPack = contextPack, + generationConfig = config.narrationGenerationConfig, + responseFormat = ResponseFormat.Text, + ) + val raw = provider.infer(inferenceRequest).text + val name = sanitizeSessionName(raw) + if (name.isBlank()) return + + val now = Clock.System.now() + eventStore.append( + NewEvent( + metadata = EventMetadata( + eventId = EventId(UUID.randomUUID().toString()), + sessionId = sessionId, + timestamp = now, + schemaVersion = 1, + causationId = null, + correlationId = null, + ), + payload = SessionNamedEvent( + sessionId = sessionId, + name = name, + timestampMs = now.toEpochMilliseconds(), + ), + ), + ) + } + + /** Strips quotes/markdown/trailing punctuation and clamps a model-produced title to one line. */ + private fun sanitizeSessionName(raw: String): String = + raw.lineSequence() + .map { it.trim() } + .firstOrNull { it.isNotBlank() } + .orEmpty() + .trim('"', '\'', '`', '*', '#', ' ', '.', ':') + .take(NAME_MAX_CHARS) + .trim() + /** * The most recent concrete tool actions in the session, read straight from the event log so * the narrator can say what actually happened (files read/written, tools run) instead of a diff --git a/core/tools/src/main/kotlin/com/correx/core/tools/process/ChildProcess.kt b/core/tools/src/main/kotlin/com/correx/core/tools/process/ChildProcess.kt new file mode 100644 index 00000000..952ba393 --- /dev/null +++ b/core/tools/src/main/kotlin/com/correx/core/tools/process/ChildProcess.kt @@ -0,0 +1,44 @@ +package com.correx.core.tools.process + +import java.io.File + +/** + * Single construction point for child processes spawned on behalf of a run — both model-authored + * shell (ShellTool) and operator-trusted gate commands (static-analysis runner) go through here. + * + * Every child is configured for *unattended* execution, closing the two ways a subprocess hangs a + * headless run: + * 1. **stdin from /dev/null** — a tool that still prompts reads EOF and fails fast instead of + * blocking on a dangling pipe until the caller's timeout kills it. + * 2. **non-interactive env overlay** — package managers / scaffolders (`npm create`, `npm init`, + * …) honour `CI`, `npm_config_yes`, etc. and take defaults rather than prompting at all. + * + * Output redirection is deliberately NOT set here: callers differ (ShellTool drains stdout/stderr + * separately; the gate runner merges them via `redirectErrorStream`). They also keep their own + * timeout backstop — this only builds the [ProcessBuilder]. + */ +object ChildProcess { + private val DEV_NULL = File("/dev/null") + + // Overlaid on the inherited environment. Forces the common Node toolchain into non-interactive + // mode so `npm create`/`npm init` and friends never block on a stdin prompt. + private val NON_INTERACTIVE_ENV = mapOf( + "CI" to "true", + "npm_config_yes" to "true", + "NPM_CONFIG_YES" to "true", + "ADBLOCK" to "1", + "DISABLE_OPENCOLLECTIVE" to "1", + ) + + /** + * A [ProcessBuilder] for [command] in [workingDir] (JVM cwd when null), with the non-interactive + * env overlay and stdin redirected from /dev/null. The caller sets any output redirection and + * owns the timeout. + */ + fun builder(command: List, workingDir: File? = null): ProcessBuilder = + ProcessBuilder(command).apply { + workingDir?.let { directory(it) } + environment().putAll(NON_INTERACTIVE_ENV) + redirectInput(ProcessBuilder.Redirect.from(DEV_NULL)) + } +} diff --git a/docs/ui/web-ui-design-spec.md b/docs/ui/web-ui-design-spec.md deleted file mode 100644 index d6b268fd..00000000 --- a/docs/ui/web-ui-design-spec.md +++ /dev/null @@ -1,21 +0,0 @@ -# Web UI Design Specification - -## Aesthetic -- **Theme**: Soft Rounded + Blue Accent. -- **Consistency**: Should align with the TUI layout plan. - -## Layout -- **Sidebar**: Centralized navigation. -- **Main Area**: Dynamic content based on selection. -- **Event Log**: Easily accessible (Source of Truth). - -## Components -- **Cards**: For task summaries, status updates. -- **Badges**: For status (e.g., `CLAIMED`, `PENDING`, `COMPLETED`). -- **Progress Bars**: For multi-step task progress. -- **Service Layer**: Placeholders for `apps/server` endpoints. - -## UX Priorities -- Clear hierarchy. -- Intuitive navigation. -- Avoid clutter (maintain information density). diff --git a/examples/workflows/prompts/architect_freestyle.md b/examples/workflows/prompts/architect_freestyle.md index 2967b877..22db3e57 100644 --- a/examples/workflows/prompts/architect_freestyle.md +++ b/examples/workflows/prompts/architect_freestyle.md @@ -79,12 +79,15 @@ Emit a JSON object that validates against the `execution_plan` schema: decompose tasks here — task creation is out of scope for the plan. - Keep stages small and single-responsibility. Prefer more stages over large monolithic prompts. -- **Scaffolding uses `file_write`, never a network installer.** A stage that sets up a project - writes `package.json`, config files, and sources directly with `file_write` — it does not - shell out to `npm create vite`, `create-react-app`, `npx`, or any `create`/`init` scaffolder. - Those fetch remote code and prompt on stdin; the run is unattended (no TTY) so they hang, and - `shell` rejects them outright. `shell` in a scaffold stage is for `npm install` / `npm run - build` on files that already exist, not for generating the project. +- **Prefer the real scaffolder over hand-writing config.** A stage that sets up a project should + use the package manager's own generator — e.g. `npm create vite@latest -- --template + react-ts` — because it produces a correct, complete, current file set (a hand-written + `package.json` reliably forgets something). Runs are unattended, but the shell forces + non-interactive mode (stdin closed, `CI=true`), so `npm create`/`npm init` no longer hang; + always pass the template/preset arg the generator requires (`-- --template react-ts`) since + there's no prompt to fall back on. `file_write` is the fallback when no generator fits the + stack. Only bare remote runners (`npx`, `bunx`, `pnpx`) stay blocked — they execute arbitrary + remote code; reach them via `npm create` instead. - **Every stage prompt must describe its exact JSON output structure when using a structured kind.** The model that runs the stage never sees the raw JSON schema — it only sees the kind's name (e.g. `analysis`, `research_report`, `review_report`) and the prompt you write diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt index 009e9374..37d3f3f4 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileEditTool.kt @@ -79,7 +79,30 @@ class FileEditTool( add(JsonPrimitive("path")) add(JsonPrimitive("operation")) }) + // Per-operation requirements. `required` above can't express these (they depend on the + // chosen operation), so a plain schema lets a constrained decoder finish a 'replace' call + // with no target/replacement — a schema-valid but useless edit the runtime then has to + // reject. The if/then conditionals make the operand params required once the operation is + // fixed, so a schema-aware decoder emits them. + put("allOf", buildJsonArray { + add(conditionalRequirement("replace", "target", "replacement")) + add(conditionalRequirement("append", "content")) + add(conditionalRequirement("patch", "patch")) + }) } + private fun conditionalRequirement(operation: String, vararg required: String): JsonObject = + buildJsonObject { + putJsonObject("if") { + putJsonObject("properties") { + putJsonObject("operation") { put("const", operation) } + } + put("required", buildJsonArray { add(JsonPrimitive("operation")) }) + } + putJsonObject("then") { + put("required", buildJsonArray { required.forEach { add(JsonPrimitive(it)) } }) + } + } + override val tier: Tier = Tier.T3 override val requiredCapabilities: Set = setOf(ToolCapability.FILE_WRITE) override val paramRoles: Map = mapOf("path" to ParamRole.PATH) @@ -232,27 +255,76 @@ class FileEditTool( val currentContent = Files.readString(path) val occurrences = currentContent.split(target).size - 1 - return if (occurrences == 1) { - val newContent = currentContent.replace(target, replacement) - Files.writeString(path, newContent) - ToolResult.Success( + return when (occurrences) { + 1 -> { + val newContent = currentContent.replace(target, replacement) + Files.writeString(path, newContent) + ToolResult.Success( + invocationId = request.invocationId, + output = "Target replaced in $pathString", + ) + } + // Don't echo the (often whole-file) target back into context — it bloats the turn and + // teaches nothing. Instead point at the closest actual line, since a miss is almost + // always whitespace/content drift the model can correct from a nudge. + 0 -> ToolResult.Failure( invocationId = request.invocationId, - output = "Target replaced in $pathString", - ) - } else { - ToolResult.Failure( - invocationId = request.invocationId, - reason = if (occurrences == 0) { - "Target '$target' not found in $pathString" - } else { - "Target '$target' found $occurrences times, expected exactly once" + reason = buildString { + append("Target not found in $pathString") + nearestLine(currentContent, target)?.let { + append(". Did you mean (whitespace/content may differ): $it ?") + } }, - // Recoverable: the model can adjust the target string and retry. + recoverable = true, + ) + else -> ToolResult.Failure( + invocationId = request.invocationId, + reason = "Target found $occurrences times in $pathString, expected exactly once — " + + "extend it with surrounding lines to make it unique.", recoverable = true, ) } } + /** + * Best-effort "did you mean" for a not-found replace target: anchor on the target's first + * non-blank line and return the file line most similar to it (levenshtein ratio), or null when + * nothing is close enough to be a useful suggestion. Mirrors ListDirTool's fuzzy directory hint. + */ + private fun nearestLine(content: String, target: String): String? { + val anchor = target.lineSequence().firstOrNull { it.isNotBlank() }?.trim()?.take(MAX_CMP) + if (anchor == null || anchor.length < MIN_ANCHOR) return null + return content.lineSequence() + .filter { it.isNotBlank() } + .map { it.trim() to similarity(anchor, it.trim().take(MAX_CMP)) } + .maxByOrNull { it.second } + ?.takeIf { it.second >= MIN_SIMILARITY } + ?.first + } + + private fun similarity(a: String, b: String): Double { + val maxLen = maxOf(a.length, b.length) + return if (maxLen == 0) 1.0 else 1.0 - levenshtein(a, b).toDouble() / maxLen + } + + private fun levenshtein(a: String, b: String): Int { + val dp = IntArray(b.length + 1) { it } + for (i in 1..a.length) { + var prev = dp[0] + dp[0] = i + for (j in 1..b.length) { + val tmp = dp[j] + dp[j] = minOf( + dp[j] + 1, + dp[j - 1] + 1, + prev + if (a[i - 1].lowercaseChar() == b[j - 1].lowercaseChar()) 0 else 1, + ) + prev = tmp + } + } + return dp[b.length] + } + private fun patch(path: Path, pathString: String, request: ToolRequest): ToolResult { val patchContent = request.parameters["patch"] as String val parentDir = path.parent ?: Paths.get(".") @@ -312,4 +384,10 @@ class FileEditTool( recoverable = false, ) } + + private companion object { + const val MIN_ANCHOR = 3 + const val MAX_CMP = 400 + const val MIN_SIMILARITY = 0.5 + } } diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileReadTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileReadTool.kt index 1e63e438..2b4d6868 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileReadTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/FileReadTool.kt @@ -93,7 +93,10 @@ class FileReadTool( return runCatching { val path = resolvePath(pathString) - if (!PathJail.isContained(path, allowedPaths)) + // Widen the jail with any operator-approved out-of-workspace paths carried on the request + // (OutsidePathAccessGrantedEvent). Reads may escape once approved; writes never see this. + val effectiveRoots = allowedPaths + request.grantedPaths.map { Paths.get(it) } + if (!PathJail.isContained(path, effectiveRoots)) ValidationResult.Invalid("Path '$pathString' is not in the allowed list") else ValidationResult.Valid diff --git a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt index 17394622..649d490e 100644 --- a/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt +++ b/infrastructure/tools/filesystem/src/main/kotlin/com/correx/infrastructure/tools/filesystem/ListDirTool.kt @@ -78,7 +78,9 @@ class ListDirTool( val pathString = (request.parameters["path"] as? String) ?: "." return runCatching { val path = resolvePath(pathString) - if (!PathJail.isContained(path, allowedPaths)) + // Operator-approved out-of-workspace paths (OutsidePathAccessGrantedEvent) widen the jail. + val effectiveRoots = allowedPaths + request.grantedPaths.map { Paths.get(it) } + if (!PathJail.isContained(path, effectiveRoots)) ValidationResult.Invalid("Path '$pathString' is not in the allowed list") else ValidationResult.Valid }.getOrElse { diff --git a/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileEditToolTest.kt b/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileEditToolTest.kt index f69ead52..58dae2c0 100644 --- a/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileEditToolTest.kt +++ b/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileEditToolTest.kt @@ -179,6 +179,28 @@ class FileEditToolTest { assertTrue(failure.reason.contains("not found")) } + @Test + fun `execute replace not-found suggests nearest line`(): Unit = runBlocking { + val tempDir = Files.createTempDirectory("file_edit_replace_suggest") + val filePath = tempDir.resolve("test.txt") + Files.writeString(filePath, "import React from 'react';\nconst greeting = 'hello world';\n") + val tool = FileEditTool(allowedPaths = setOf(tempDir)) + val request = createRequest( + mapOf( + "operation" to "replace", + "path" to filePath.toString(), + // whitespace/quote drift from the real line — the classic near-miss + "target" to "const greeting = \"hello world\"", + "replacement" to "const greeting = 'hi';", + ), + ) + + val failure = tool.execute(request) as ToolResult.Failure + assertTrue(failure.reason.contains("not found")) + assertTrue(failure.reason.contains("Did you mean")) + assertTrue(failure.reason.contains("const greeting = 'hello world';")) + } + @Test fun `execute replace failure multiple matches`(): Unit = runBlocking { val tempDir = Files.createTempDirectory("file_edit_replace_fail_multi") diff --git a/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileReadToolTest.kt b/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileReadToolTest.kt index b477202a..1110784e 100644 --- a/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileReadToolTest.kt +++ b/infrastructure/tools/filesystem/src/test/kotlin/com/correx/infrastructure/tools/filesystem/FileReadToolTest.kt @@ -49,6 +49,21 @@ class FileReadToolTest { assertTrue((result as ValidationResult.Invalid).reason.contains("not in the allowed list")) } + @Test + fun `validateRequest admits an out-of-jail path carried on request grantedPaths`(): Unit = runBlocking { + // A file the static jail does NOT cover (an unrelated temp dir), mirroring an out-of-workspace + // reference the operator approved. Without the grant it's rejected; with it, admitted. + val jailDir = Files.createTempDirectory("jail") + val outside = Files.createTempFile("outside_ref", ".md") + val tool = FileReadTool(allowedPaths = setOf(jailDir)) + + val ungranted = createRequest(outside.toString()) + assertTrue(tool.validateRequest(ungranted) is ValidationResult.Invalid) + + val granted = createRequest(outside.toString()).copy(grantedPaths = setOf(outside.toString())) + assertEquals(ValidationResult.Valid, tool.validateRequest(granted)) + } + @Test fun `validateRequest returns Invalid for missing path parameter`(): Unit = runBlocking { val tool = FileReadTool() diff --git a/infrastructure/tools/src/main/kotlin/com/correx/infrastructure/tools/shell/ShellTool.kt b/infrastructure/tools/src/main/kotlin/com/correx/infrastructure/tools/shell/ShellTool.kt index 192ae6d9..b01d0d1d 100644 --- a/infrastructure/tools/src/main/kotlin/com/correx/infrastructure/tools/shell/ShellTool.kt +++ b/infrastructure/tools/src/main/kotlin/com/correx/infrastructure/tools/shell/ShellTool.kt @@ -14,6 +14,7 @@ import com.correx.core.tools.contract.ToolCapability import com.correx.core.tools.contract.ToolExecutor import com.correx.core.tools.contract.ToolResult import com.correx.core.tools.contract.ValidationResult +import com.correx.core.tools.process.ChildProcess import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.TimeoutCancellationException import kotlinx.coroutines.async @@ -83,28 +84,20 @@ class ShellTool( } } - // Return a rejection reason if argv is a fetch-and-execute-remote-code or interactive-scaffolder - // invocation, else null. Matched by shape, not just argv[0], because the danger for package - // managers is the *subcommand* (`npm install` is fine; `npm create`/`npm init` scaffolds and - // prompts). Deterministic, allowlist-independent — see [validateRequest]. + // Return a rejection reason if argv fetches and runs an arbitrary *remote* package in one shot + // (npx/bunx/pnpx), else null. That's remote code execution — a distinct threat we still deny. + // Package-manager scaffolders (`npm create`, `npm init`, `npm install`) are NO LONGER blocked: + // the hang they used to guard against (stdin prompt with no TTY) is now closed at the process + // factory — every child gets stdin from /dev/null and a non-interactive env overlay + // ([ChildProcess]). Matched by shape, not just argv[0], because a runner can hide inside a shell + // command line run via `sh -c` (`cd web && npx create-foo`), where argv[0] is `cd`. private fun deniedReason(argv: List): String? { - // Scan every token, not just argv[0]: a scaffolder can hide inside a shell command line that - // runs via `sh -c` (`cd web && npm create vite`), where argv[0] is `cd`. + // Scan every token, not just argv[0]: a runner can hide behind a shell builtin at argv[0]. val progs = argv.map { it.substringAfterLast('/') } - val redirect = "Scaffold by writing files directly with file_write (package.json, configs, " + - "sources) — do not shell out to a network installer; it needs a TTY and will hang unattended." - val runner = progs.firstOrNull { it in REMOTE_EXEC_RUNNERS } - val scaffolder = progs.firstOrNull { it.startsWith("create-") } - val pmIdx = progs.indexOfFirst { it in PACKAGE_MANAGERS } - return when { - runner != null -> - "'$runner' fetches and runs remote code with no TTY, which hangs an unattended run. $redirect" - scaffolder != null -> - "'$scaffolder' is an interactive project scaffolder that prompts on stdin. $redirect" - pmIdx >= 0 && progs.getOrNull(pmIdx + 1) in SCAFFOLD_SUBCOMMANDS -> - "'${progs[pmIdx]} ${progs[pmIdx + 1]}' scaffolds a project interactively " + - "(prompts on stdin). $redirect" - else -> null + return progs.firstOrNull { it in REMOTE_EXEC_RUNNERS }?.let { runner -> + "'$runner' downloads and runs an arbitrary remote package in one shot (remote code " + + "execution). Use the package manager's own scaffold command instead — e.g. " + + "`npm create vite@latest -- --template react-ts` — or write the files with file_write." } } @@ -170,7 +163,7 @@ class ShellTool( // shell API. A still-unrunnable program returns RECOVERABLE (model adapts, not fatal). val command = if (looksLikeShellCommand(argv)) listOf("sh", "-c", argv.joinToString(" ")) else argv val process = runCatching { - ProcessBuilder(command).apply { workingDir?.let { directory(it.toFile()) } }.start() + ChildProcess.builder(command, workingDir?.toFile()).start() }.getOrElse { return@let ToolResult.Failure( invocationId = request.invocationId, @@ -213,10 +206,29 @@ class ShellTool( val SHELL_OPERATORS = setOf("&&", "||", "|", ">", ">>", "<", ";", "&", "2>", "2>&1") // Runners that download and execute an arbitrary remote package in one shot. val REMOTE_EXEC_RUNNERS = setOf("npx", "bunx", "pnpx") - val PACKAGE_MANAGERS = setOf("npm", "yarn", "pnpm", "bun") - // Package-manager subcommands that generate a project by pulling+running a remote scaffolder - // and prompting on stdin (npm create vite, yarn create next-app, pnpm dlx, …). - val SCAFFOLD_SUBCOMMANDS = setOf("create", "init", "dlx") + } + + // Combines a failed process's stdout and stderr into one labelled, head/tail-bounded block for + // the model's failure feedback. Labels each stream (only when non-blank) so the model can tell + // compiler diagnostics on stdout from runner errors on stderr; the same HEAD/TAIL window as the + // success-path compressor keeps a large build log from flooding the next context window. + private fun boundFailureOutput(stdout: String, stderr: String): String { + val sections = buildList { + if (stdout.isNotBlank()) add("stdout:\n${stdout.trimEnd()}") + if (stderr.isNotBlank()) add("stderr:\n${stderr.trimEnd()}") + } + return boundLines(sections.joinToString("\n\n")) + } + + // Keeps the first HEAD_LINES and last TAIL_LINES of a multi-line block, dropping the middle with + // an elision marker — build errors cluster at both ends (first failures, final summary), so a + // head+tail window preserves the actionable lines while bounding size. + private fun boundLines(text: String): String { + val lines = text.lines() + if (lines.size <= HEAD_LINES + TAIL_LINES) return text + val omitted = lines.size - HEAD_LINES - TAIL_LINES + return (lines.take(HEAD_LINES) + "… ($omitted lines omitted) …" + lines.takeLast(TAIL_LINES)) + .joinToString("\n") } private suspend fun runCmd(request: ToolRequest, process: Process): ToolResult = withTimeout(timeoutMs) { @@ -230,11 +242,18 @@ class ShellTool( val metadata = if (stderr.isNotBlank()) mapOf("stderr" to stderr) else emptyMap() if (exitCode != 0) { // A non-zero exit is normal command behaviour (missing path, no grep match, git - // --exit-code, test, diff). Surface it recoverably so the model sees the code + stderr + // --exit-code, test, diff). Surface it recoverably so the model sees the code + output // and adapts, rather than aborting the stage on the first failed probe. + // + // BOTH streams, not just stderr: compilers and test runners (tsc, vite, jest, pytest…) + // print their diagnostics to STDOUT, so a stderr-only failure message returns a bare + // "exited with code 2" and the model fixes blind — it can see the build failed but not + // which file/line. Combine stdout+stderr, head/tail-bounded so a large build log can't + // flood the next context window. + val detail = boundFailureOutput(stdout, stderr) ToolResult.Failure( invocationId = request.invocationId, - reason = "Process exited with code $exitCode${if (stderr.isNotBlank()) ": $stderr" else ""}", + reason = "Process exited with code $exitCode${if (detail.isNotBlank()) ":\n$detail" else ""}", recoverable = true, ) } else { diff --git a/infrastructure/tools/src/test/kotlin/com/correx/infrastructure/tools/shell/ShellToolTest.kt b/infrastructure/tools/src/test/kotlin/com/correx/infrastructure/tools/shell/ShellToolTest.kt index 1aa10f29..14dd79ed 100644 --- a/infrastructure/tools/src/test/kotlin/com/correx/infrastructure/tools/shell/ShellToolTest.kt +++ b/infrastructure/tools/src/test/kotlin/com/correx/infrastructure/tools/shell/ShellToolTest.kt @@ -102,6 +102,24 @@ class ShellToolTest { assertTrue(failure.recoverable, "non-zero exit must be recoverable so the model can adapt") } + @Test + fun `execute surfaces stdout in the failure reason for a non-zero exit`(): Unit = runBlocking { + // Compilers/test runners (tsc, vite, jest) print diagnostics to STDOUT and exit non-zero. + // The failure reason must carry that output, else the model sees only "exited with code 2" + // and fixes blind. Regression for session 6f9a9f08 recovery loop. + val tool = ShellTool() + val request = createRequest(listOf("sh", "-c", "echo 'hooks.ts(57,3): error TS1005'; exit 2")) + val result = tool.execute(request) + + assertTrue(result is ToolResult.Failure) + val failure = result as ToolResult.Failure + assertTrue(failure.reason.contains("exited with code 2")) + assertTrue( + failure.reason.contains("TS1005"), + "stdout diagnostics must reach the model's failure feedback, got: ${failure.reason}", + ) + } + @Test fun `execute returns Failure on timeout`(): Unit = runBlocking { // Using sleep command to simulate timeout @@ -182,15 +200,12 @@ class ShellToolTest { } @Test - fun `denylist blocks npm create even in allow-all mode`() { - // Repro: gemma ran `npm create vite@latest` — a network scaffolder that prompts on stdin. - // Empty allowedExecutables = allow-all, yet this must still be rejected. + fun `denylist allows npm create — hang closed at the process factory`() { + // `npm create vite@latest` used to be blocked as an interactive scaffolder. The stdin-prompt + // hang is now closed at ChildProcess (stdin=/dev/null + non-interactive env), so the idiomatic + // scaffold path is allowed again. Empty allowedExecutables = allow-all. val tool = ShellTool(allowedExecutables = emptySet()) - val result = tool.validateRequest(createRequest(listOf("npm", "create", "vite"))) - assertTrue(result is ValidationResult.Invalid) - val reason = (result as ValidationResult.Invalid).reason - assertTrue(reason.contains("npm create"), reason) - assertTrue(reason.contains("file_write"), reason) + assertEquals(ValidationResult.Valid, tool.validateRequest(createRequest(listOf("npm", "create", "vite")))) } @Test @@ -201,32 +216,24 @@ class ShellToolTest { } @Test - fun `denylist blocks create- scaffolder run directly`() { - val result = ShellTool().validateRequest(createRequest(listOf("create-next-app", "app"))) + fun `denylist catches remote runner hidden in a shell command line`() { + // argv[0] is `cd`, but `npx` rides along and would run via sh -c. + val result = ShellTool().validateRequest(createRequest(listOf("cd", "web", "&&", "npx", "create-foo"))) assertTrue(result is ValidationResult.Invalid) - assertTrue((result as ValidationResult.Invalid).reason.contains("scaffolder"), result.reason) - } - - @Test - fun `denylist catches scaffolder hidden in a shell command line`() { - // argv[0] is `cd`, but `npm create` rides along and would run via sh -c. - val result = ShellTool().validateRequest(createRequest(listOf("cd", "web", "&&", "npm", "create", "vite"))) - assertTrue(result is ValidationResult.Invalid) - assertTrue((result as ValidationResult.Invalid).reason.contains("npm create"), result.reason) + assertTrue((result as ValidationResult.Invalid).reason.contains("remote code"), result.reason) } @Test fun `denylist allows npm install`() { - // The subcommand is what's dangerous — `npm install` on existing files is fine. assertEquals(ValidationResult.Valid, ShellTool().validateRequest(createRequest(listOf("npm", "install")))) } @Test - fun `denylist takes precedence over the allowlist`() { - // Even if npm is explicitly allowed, `npm create` stays blocked. - val tool = ShellTool(allowedExecutables = setOf("npm")) - val result = tool.validateRequest(createRequest(listOf("npm", "create", "vite"))) + fun `denylist takes precedence over the allowlist for remote runners`() { + // Even if npx is explicitly allowed, remote-exec stays blocked. + val tool = ShellTool(allowedExecutables = setOf("npx")) + val result = tool.validateRequest(createRequest(listOf("npx", "create-react-app", "app"))) assertTrue(result is ValidationResult.Invalid) - assertTrue((result as ValidationResult.Invalid).reason.contains("file_write"), result.reason) + assertTrue((result as ValidationResult.Invalid).reason.contains("remote code"), result.reason) } } diff --git a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/ExecutionPlanCompiler.kt b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/ExecutionPlanCompiler.kt index 0653258f..68fa7198 100644 --- a/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/ExecutionPlanCompiler.kt +++ b/infrastructure/workflow/src/main/kotlin/com/correx/infrastructure/workflow/ExecutionPlanCompiler.kt @@ -27,9 +27,25 @@ private const val RECOVERY_PROMPT = "You are the recovery stage. An upstream stage failed a verification gate it could not fix " + "(it lacked write access). The failure ticket — the failing stage, the gate, and its exact " + "output — is in your context under \"## Recovery ticket\". Read the gate output, inspect the " + - "relevant files (file_read / list_dir), and make the minimal edits that fix the reported " + - "cause. Do not re-scaffold or rewrite working code. When done, control returns to the stage " + - "that failed so its gate re-runs." + "relevant files (file_read / list_dir), then FIX the reported cause. You MUST make the " + + "change before finishing — investigation alone does not resolve the ticket, and handing " + + "back without an edit only re-triggers the same failure.\n" + + "Fix the actual cause the gate reports, whatever form it takes:\n" + + "- Broken code / config → edit it with file_write.\n" + + "- A MISSING file the build needs (e.g. package.json, tsconfig.json, a build/entry file " + + "the gate can't find) → produce it. A build gate failing with \"no such file\" is fixed by " + + "creating that file. If a whole project skeleton is missing, prefer the package manager's " + + "own generator via shell — e.g. `npm create vite@latest -- --template react-ts` — " + + "which writes a correct, complete file set (always pass the template/preset arg it " + + "requires; the run is non-interactive so there's no prompt to fall back on). Fall back to " + + "writing the files directly with file_write when no generator fits. `npm install` is " + + "available to fetch deps once package.json exists. Only bare remote runners (`npx`, " + + "`bunx`, `pnpx`) are blocked; use `npm create` instead.\n" + + "Do not gratuitously rewrite code that already works — touch only what the gate blames. But " + + "\"minimal\" means minimal to actually pass the gate, not zero.\n" + + "When the cause is repaired, finish with a short resolution summary that mirrors the ticket: " + + "the failing stage and gate, the root cause you found, and the exact files created/edited " + + "and commands run. Then control returns to the stage that failed so its gate re-runs." // Freestyle plans don't specify a per-stage token budget, so without this every compiled stage // inherits StageConfig's 4096 default — 1/8th of what the static role_pipeline execution stages @@ -118,7 +134,13 @@ class ExecutionPlanCompiler( StageId(s.id) to StageConfig( produces = listOf(TypedArtifactSlot(name = ArtifactId(s.produces), kind = kind)), needs = s.needs.map { ArtifactId(it) }.toSet(), - allowedTools = s.tools.toSet(), + // Every stage gets the full registered tool universe. The architect's per-stage + // `tools` pick was advisory and mis-scoped stages (e.g. an edit-shaped stage granted + // file_write but not file_edit → forced full-file rewrites; a repair stage unable to + // touch a file it must fix). writeManifest still jails WHICH paths a stage may write, + // so widening the tool grant does not widen write authority. Falls back to the + // architect's pick only when no universe was supplied (bare-constructor unit tests). + allowedTools = knownTools.ifEmpty { s.tools.toSet() }, writeManifest = writeManifest, // Option A: the concrete (non-glob) subset of the declared writes are expected files // the contract gate will require to exist. Globs (src/**) are write-permission scopes, @@ -144,7 +166,7 @@ class ExecutionPlanCompiler( null } else { StageId(RECOVERY_STAGE) to StageConfig( - allowedTools = setOf("file_write", "shell"), + allowedTools = knownTools.ifEmpty { setOf("file_write", "file_edit", "shell") }, tokenBudget = DEFAULT_STAGE_TOKEN_BUDGET, generationConfig = DEFAULT_STAGE_GENERATION, metadata = mapOf("role" to "recovery", "promptInline" to RECOVERY_PROMPT), diff --git a/testing/deterministic/src/test/kotlin/TalkieFacadeTest.kt b/testing/deterministic/src/test/kotlin/TalkieFacadeTest.kt index d768a5a0..eef75c6a 100644 --- a/testing/deterministic/src/test/kotlin/TalkieFacadeTest.kt +++ b/testing/deterministic/src/test/kotlin/TalkieFacadeTest.kt @@ -762,6 +762,8 @@ class TalkieFacadeTest { impl.onUserInput(sessionId = sessionId, input = input, mode = chatMode) override suspend fun narrate(sessionId: SessionId, trigger: com.correx.core.talkie.model.NarrationTrigger) = impl.narrate(sessionId = sessionId, trigger = trigger) + override suspend fun nameSession(sessionId: SessionId, intent: String) = + impl.nameSession(sessionId = sessionId, intent = intent) } } diff --git a/testing/integration/src/test/kotlin/RecoveryRoutingTest.kt b/testing/integration/src/test/kotlin/RecoveryRoutingTest.kt index eb756fd8..edfebf96 100644 --- a/testing/integration/src/test/kotlin/RecoveryRoutingTest.kt +++ b/testing/integration/src/test/kotlin/RecoveryRoutingTest.kt @@ -53,7 +53,11 @@ import com.correx.core.sessions.projections.replay.DefaultEventReplayer import com.correx.core.sessions.projections.replay.EventReplayer import com.correx.core.toolintent.WorkspacePolicy import com.correx.core.tools.contract.Tool +import com.correx.core.tools.contract.ToolExecutor +import com.correx.core.tools.contract.ToolResult +import com.correx.core.events.events.ToolRequest import com.correx.core.tools.registry.ToolRegistry +import com.correx.infrastructure.tools.filesystem.FileWriteTool import com.correx.core.transitions.graph.StageConfig import com.correx.core.transitions.graph.TransitionEdge import com.correx.core.transitions.graph.WorkflowGraph @@ -282,7 +286,282 @@ class RecoveryRoutingTest { assertTrue(tickets.map { it.fingerprint }.toSet().size == tickets.size, "each route recorded a distinct fingerprint") } + /** Emits a tool call on odd infers, ends the round on even — but the tool depends on the stage: + * `impl` writes a file (so it becomes the file's author), any other stage runs `shell true`. */ + private class StageRoutingProvider : InferenceProvider { + override val id = ProviderId("stage-router") + override val name = "stage-router" + override val tokenizer: Tokenizer = com.correx.testing.fixtures.inference.MockTokenizer() + private val counts = mutableMapOf() + override suspend fun infer(request: InferenceRequest): InferenceResponse { + val stage = request.stageId.value + val n = (counts[stage] ?: 0) + 1 + counts[stage] = n + if (n % 2 == 0) { + return InferenceResponse( + requestId = request.requestId, text = "done", finishReason = FinishReason.Stop, + tokensUsed = TokenUsage(10, 5), latencyMs = 0, + ) + } + val (tool, args) = if (stage == "impl") { + "file_write" to """{"path":"App.tsx","content":"export const x = 1;\n"}""" + } else { + "shell" to """{"argv":["true"]}""" + } + return InferenceResponse( + requestId = request.requestId, text = "", finishReason = FinishReason.ToolCall, + tokensUsed = TokenUsage(10, 5), latencyMs = 0, + toolCalls = listOf( + ToolCallRequest(id = "tc-$stage-$n", function = ToolCallFunction(name = tool, arguments = args)), + ), + ) + } + override suspend fun healthCheck(): ProviderHealth = ProviderHealth.Healthy + override fun capabilities(): Set = setOf(CapabilityScore(ModelCapability.General, 1.0)) + } + + /** Like [StageRoutingProvider], but records the context-entry sourceTypes it saw on each `impl` + * inference — so a test can assert whether impl's own prompt (agentPrompt) was present. */ + private class RecordingRoutingProvider : InferenceProvider { + override val id = ProviderId("recording-router") + override val name = "recording-router" + override val tokenizer: Tokenizer = com.correx.testing.fixtures.inference.MockTokenizer() + val implSourceTypes = mutableListOf>() + private val counts = mutableMapOf() + override suspend fun infer(request: InferenceRequest): InferenceResponse { + val stage = request.stageId.value + if (stage == "impl") { + implSourceTypes.add(request.contextPack.layers.values.flatten().map { it.sourceType }.toSet()) + } + val n = (counts[stage] ?: 0) + 1 + counts[stage] = n + if (n % 2 == 0) { + return InferenceResponse( + requestId = request.requestId, text = "done", finishReason = FinishReason.Stop, + tokensUsed = TokenUsage(10, 5), latencyMs = 0, + ) + } + val (tool, args) = if (stage == "impl") { + "file_write" to """{"path":"App.tsx","content":"export const x = 1;\n"}""" + } else { + "shell" to """{"argv":["true"]}""" + } + return InferenceResponse( + requestId = request.requestId, text = "", finishReason = FinishReason.ToolCall, + tokensUsed = TokenUsage(10, 5), latencyMs = 0, + toolCalls = listOf( + ToolCallRequest(id = "tc-$stage-$n", function = ToolCallFunction(name = tool, arguments = args)), + ), + ) + } + override suspend fun healthCheck(): ProviderHealth = ProviderHealth.Healthy + override fun capabilities(): Set = setOf(CapabilityScore(ModelCapability.General, 1.0)) + } + + private class MultiToolRegistry(private val tools: List) : ToolRegistry { + override fun resolve(name: String): Tool? = tools.firstOrNull { it.name == name } + override fun all(): List = tools + } + + // impl (file_write) --t-impl--> verify (write-less, failing static gate naming impl's file) --> done + // No `recovery` stage exists: the ticket must route to the file's AUTHOR (impl) via route-to-owner, + // not to a generic recovery stage. impl repairs and returns to verify (ticketReturnMove). + private fun ownerGraph(): WorkflowGraph = WorkflowGraph( + id = "route-to-owner-test", + stages = mapOf( + StageId("impl") to StageConfig(allowedTools = setOf("file_write")), + StageId("verify") to StageConfig(allowedTools = setOf("shell"), staticAnalysis = listOf("typecheck")), + ), + transitions = setOf( + TransitionEdge(TransitionId("t-impl"), StageId("impl"), StageId("verify"), condition = { true }), + TransitionEdge(TransitionId("t-verify"), StageId("verify"), StageId("done"), condition = { true }), + ), + start = StageId("impl"), + ) + + // impl (file_write) + verify + a recovery stage (role=recovery, holds file_write). Tier 1 routes to + // the author (impl); when its budget exhausts, tier 2 escalates to the arbiter (recovery). + private fun ownerAndArbiterGraph(): WorkflowGraph = WorkflowGraph( + id = "route-to-owner-escalation-test", + stages = mapOf( + StageId("impl") to StageConfig(allowedTools = setOf("file_write")), + StageId("verify") to StageConfig(allowedTools = setOf("shell"), staticAnalysis = listOf("typecheck")), + StageId("recovery") to StageConfig( + allowedTools = setOf("shell", "file_write"), + metadata = mapOf("role" to "recovery"), + ), + ), + transitions = setOf( + TransitionEdge(TransitionId("t-impl"), StageId("impl"), StageId("verify"), condition = { true }), + TransitionEdge(TransitionId("t-verify"), StageId("verify"), StageId("done"), condition = { true }), + ), + start = StageId("impl"), + ) + + /** Builds an orchestrator whose only writing stage is `impl` (via file_write) and whose `verify` + * static gate fails with [staticOutput] — so route-to-owner can resolve `impl` as the file author. */ + private fun buildMultiToolOrchestrator( + staticOutput: String = "App.tsx: TS2322 type mismatch", + provider: InferenceProvider = StageRoutingProvider(), + ): Pair { + val eventStore = InMemoryEventStore() + val shell = ShellTool(allowedExecutables = setOf("true"), workingDir = workspace) + val fileWrite = FileWriteTool(allowedPaths = setOf(workspace), workingDir = workspace) + val toolRegistry = MultiToolRegistry(listOf(shell, fileWrite)) + // Dispatch delegate: SandboxedToolExecutor resolves the tool from the registry, then hands the + // request to this delegate — which routes to the matching tool by name. + val dispatch = object : ToolExecutor { + override suspend fun execute(request: ToolRequest): ToolResult = + (toolRegistry.resolve(request.toolName) as ToolExecutor).execute(request) + } + val executor = SandboxedToolExecutor( + delegate = dispatch, registry = toolRegistry, eventDispatcher = EventDispatcher(eventStore), + workDir = workspace, artifactStore = NoopArtifactStore(), + ) + val inferenceRouter = object : InferenceRouter { + override suspend fun route(stageId: StageId, requiredCapabilities: Set) = provider + } + val repositories = OrchestratorRepositories( + eventStore = eventStore, + inferenceRepository = InferenceRepository(object : EventReplayer { + override fun rebuild(sessionId: SessionId) = InferenceState() + }), + orchestrationRepository = OrchestrationRepository( + DefaultEventReplayer(eventStore, OrchestrationProjector(DefaultOrchestrationReducer())), + ), + sessionRepository = DefaultSessionRepository( + DefaultEventReplayer(eventStore, SessionProjector(DefaultSessionReducer())), + ), + artifactRepository = com.correx.infrastructure.persistence.artifact.LiveArtifactRepository( + eventStore, DefaultArtifactReducer(), + ), + approvalRepository = DefaultApprovalRepository( + DefaultEventReplayer(eventStore, ApprovalProjector(DefaultApprovalReducer())), + ), + ) + val engines = OrchestratorEngines( + transitionResolver = DefaultTransitionResolver { _, _ -> true }, + contextPackBuilder = ContextFixtures.simpleBuilder(), + inferenceRouter = inferenceRouter, + validationPipeline = ValidationPipeline(validators = listOf(cyclePolicyMissingValidator())), + riskAssessor = DefaultRiskAssessor(), + toolExecutor = executor, + toolRegistry = toolRegistry, + workspacePolicy = WorkspacePolicy(workspace), + // Gate names the file impl wrote, so route-to-owner can resolve the author from the manifest. + staticAnalysisRunner = StaticAnalysisRunner { _, _ -> + StaticAnalysisRunResult(exitCode = 1, output = staticOutput) + }, + approvalEngine = object : ApprovalEngine { + override fun evaluate( + request: DomainApprovalRequest, + context: ApprovalContext, + grants: List, + now: Instant, + ) = ApprovalDecision( + id = null, requestId = request.id, outcome = ApprovalOutcome.AUTO_APPROVED, + state = ApprovalStatus.COMPLETED, tier = request.tier, contextSnapshot = context, + resolutionTimestamp = now, reason = "test-auto-approve", + ) + }, + ) + val orchestrator = DefaultSessionOrchestrator( + repositories = repositories, + engines = engines, + retryCoordinator = DefaultRetryCoordinator(eventStore), + artifactStore = NoopArtifactStore(), + decisionJournalRepository = DefaultDecisionJournalRepository( + DefaultEventReplayer(eventStore, DecisionJournalProjector(DefaultDecisionJournalReducer())), + ), + ) + return orchestrator to eventStore + } + + @Test + fun `write-less gate failure routes the ticket to the file's author (route-to-owner)`(): Unit = runBlocking { + val (orchestrator, eventStore) = buildMultiToolOrchestrator() + val sessionId = SessionId("route-to-owner") + val config = OrchestrationConfig(retryPolicy = RetryPolicy(maxAttempts = 10, backoffMs = 0)) + + val result = orchestrator.run(sessionId, ownerGraph(), config) + + assertTrue(result is WorkflowResult.Failed, "route budget still exhausts terminally") + val tickets = eventStore.read(sessionId).mapNotNull { it.payload as? FailureTicketOpenedEvent } + assertTrue(tickets.isNotEmpty(), "verify's write-less gate failure must open a ticket") + tickets.forEach { + assertEquals("verify", it.stageId.value, "the gate that failed is verify") + assertEquals("impl", it.routeTo.value, "ticket routes to the file's author, not a recovery stage") + assertEquals("file_write", it.requiredCapability) + assertTrue(!it.escalated, "no recovery stage exists, so all routes stay tier-1 (owner)") + } + } + + @Test + fun `owner budget exhaustion escalates the ticket to the arbiter (intent-holder)`(): Unit = runBlocking { + // Same constant failure every round, so neither tier makes progress: the owner (impl) burns its + // RECOVERY_ROUTE_BUDGET, THEN the ticket escalates to the arbiter (recovery) for INTENT_ROUTE_BUDGET + // more routes, THEN terminal. Proves the two-tier ladder: author first, intent-holder as escalation. + val (orchestrator, eventStore) = buildMultiToolOrchestrator() + val sessionId = SessionId("owner-escalation") + val config = OrchestrationConfig(retryPolicy = RetryPolicy(maxAttempts = 20, backoffMs = 0)) + + val result = orchestrator.run(sessionId, ownerAndArbiterGraph(), config) + + assertTrue(result is WorkflowResult.Failed, "ladder still exhausts terminally") + val tickets = eventStore.read(sessionId).mapNotNull { it.payload as? FailureTicketOpenedEvent } + val tier1 = tickets.filter { !it.escalated } + val tier2 = tickets.filter { it.escalated } + assertEquals(RECOVERY_ROUTE_BUDGET_TEST, tier1.size, "tier-1 spends the owner budget first") + tier1.forEach { assertEquals("impl", it.routeTo.value, "tier-1 routes to the file's author") } + assertEquals(INTENT_ROUTE_BUDGET_TEST, tier2.size, "then tier-2 spends the arbiter budget") + tier2.forEach { assertEquals("recovery", it.routeTo.value, "tier-2 escalates to the arbiter") } + // ordering: every tier-1 ticket precedes every tier-2 ticket (escalation only after owner exhaust) + val firstEscalatedIdx = tickets.indexOfFirst { it.escalated } + assertTrue(tickets.take(firstEscalatedIdx).none { it.escalated }, "owner routes come before escalation") + } + + // impl carries a generative "scaffold" mandate as its own prompt. On the FORWARD visit it applies; + // when the ticket routes impl back to repair its own file, that mandate must be suppressed (else it + // re-scaffolds and stubs the real file). Same topology as ownerGraph otherwise. + private fun ownerGraphWithMandate(): WorkflowGraph = WorkflowGraph( + id = "owner-mandate-suppression-test", + stages = mapOf( + StageId("impl") to StageConfig( + allowedTools = setOf("file_write"), + metadata = mapOf("promptInline" to "Scaffold the entire project by creating placeholder files."), + ), + StageId("verify") to StageConfig(allowedTools = setOf("shell"), staticAnalysis = listOf("typecheck")), + ), + transitions = setOf( + TransitionEdge(TransitionId("t-impl"), StageId("impl"), StageId("verify"), condition = { true }), + TransitionEdge(TransitionId("t-verify"), StageId("verify"), StageId("done"), condition = { true }), + ), + start = StageId("impl"), + ) + + @Test + fun `ticket re-entry suppresses the owner stage's own generative mandate`(): Unit = runBlocking { + val provider = RecordingRoutingProvider() + val (orchestrator, _) = buildMultiToolOrchestrator(provider = provider) + val sessionId = SessionId("mandate-suppression") + val config = OrchestrationConfig(retryPolicy = RetryPolicy(maxAttempts = 10, backoffMs = 0)) + + orchestrator.run(sessionId, ownerGraphWithMandate(), config) + + val impl = provider.implSourceTypes + assertTrue(impl.isNotEmpty(), "impl must have run") + // Forward visit: impl's own prompt is present, no ticket yet. + assertTrue(impl.first().contains("agentPrompt"), "forward visit keeps the stage's own prompt") + assertTrue("recoveryTicket" !in impl.first(), "no recovery ticket before any failure") + // Ticket re-entry: the generative mandate is gone, the recovery ticket is the sole mandate. + assertTrue( + impl.any { "recoveryTicket" in it && "agentPrompt" !in it }, + "a repair re-entry dropped the generative mandate and carried only the recovery ticket; saw $impl", + ) + } + private companion object { + const val INTENT_ROUTE_BUDGET_TEST = 2 // Mirror of DefaultSessionOrchestrator.RECOVERY_ROUTE_BUDGET (private there); kept in sync by the // `[1, 2]` assertion in the exhaustion test above. const val RECOVERY_ROUTE_BUDGET_TEST = 2