Files
correx/examples/workflows/role_pipeline.toml
T
kami 365eb8a279 feat(kernel): static-first reviewer gate (role-reliability §5)
Run operator-configured static-analysis commands (compiler/detekt/formatters) as a
deterministic harness step on the producing stage, before the LLM reviewer. A stage
declares `static_analysis = [commands]`; after it produces its artifact, each command
runs in the workspace root and the results are recorded in a StaticAnalysisCompletedEvent
(invariant #9 env observation — recorded live, folded on replay, never re-run). A
non-clean command fails the stage retryably with its output fed back verbatim (§2:
compiler/test output is ground truth), so the implementer fixes it and only static-clean
code ever reaches the reviewer. This makes §5's "reviewer context excludes static findings"
mechanical — the findings are resolved upstream, so the reviewer can't waste inference on
what detekt catches for free; no output-parsing, no context plumbing.

- StaticAnalysisCompletedEvent + StaticAnalysisFinding (core:events) + registration.
- StaticAnalysisRunner seam + ProcessStaticAnalysisRunner (whitespace-split argv, stderr
  merged, timeout→nonzero exit) in core:kernel; wired (nullable) into OrchestratorEngines
  and the server. Null runner / no workspace root → logged no-op, never blocks the run.
- StageConfig.staticAnalysis + `static_analysis` TOML field; runStaticAnalysis folded into
  a runPostStageGates sequence (produces → grounding → echo → static analysis).
- Documented `static_analysis` on the role_pipeline implementer stage (commented; commands
  are workspace-specific). The reviewer prompt half shipped in 3467826.
2026-06-14 23:51:49 +04:00

164 lines
5.8 KiB
TOML

# Role pipeline: analyst → architect → brief_echo → planner → implementer ⇄ reviewer
#
# Each stage produces a typed artifact that the next stage `needs`, so work flows forward
# without a human relaying notes. The decision journal (pinned into every stage's context)
# carries steering/approvals/verdicts across the whole run, so the reviewer sees the same
# ground truth as the planner.
#
# The implementer⇄reviewer loop is gated by review_report.verdict:
# approved → done
# changes_requested → back to implementer (capped by implementer.max_retries, then escalates)
#
# Requires these artifact kinds in ~/.config/correx/config.toml (schemas under docs/schemas/):
# [[artifacts]]
# id = "analysis"; schema_path = "schemas/analysis.json"; llm_emitted = true
# [[artifacts]]
# id = "design"; schema_path = "schemas/design.json"; llm_emitted = true
# [[artifacts]]
# id = "impl_plan"; schema_path = "schemas/impl_plan.json"; llm_emitted = true
# [[artifacts]]
# id = "review_report"; schema_path = "schemas/review_report.json"; llm_emitted = true
# [[artifacts]]
# id = "brief_echo"; schema_path = "schemas/brief_echo.json"; llm_emitted = true
#
# Prompt files (prompts/*.md, relative to this workflow) must exist for a real run.
id = "role_pipeline"
start = "analyst"
# 1. Understand the request and the relevant code. Read-only.
# ground_references: every workspace-relative file path the analysis names is checked
# for existence; a hallucinated path fails the stage and retries with the misses fed back.
[[stages]]
id = "analyst"
prompt = "prompts/analyst.md"
produces = [{ name = "analysis", kind = "analysis" }]
allowed_tools = ["file_read", "ShellTool"]
ground_references = true
token_budget = 16384
max_retries = 2
# 2. Decide the approach and component boundaries.
[[stages]]
id = "architect"
prompt = "prompts/architect.md"
needs = ["analysis"]
produces = [{ name = "design", kind = "design" }]
token_budget = 16384
max_retries = 2
# 3a. Before planning: echo the analyst brief back as a structured artifact.
# The brief_echo gate diffs the restatement against the original analysis;
# on divergence (dropped requirements or hallucinated files) it fails the
# stage (retryable) so the pipeline cannot reach plan generation on a misread brief.
[[stages]]
id = "brief_echo"
prompt = "prompts/brief_echo.md"
needs = ["analysis"]
produces = [{ name = "brief_echo", kind = "brief_echo" }]
brief_echo = true
token_budget = 8192
max_retries = 2
# 3b. Break the design into ordered, verifiable steps.
[[stages]]
id = "planner"
prompt = "prompts/planner.md"
needs = ["design"]
produces = [{ name = "impl_plan", kind = "impl_plan" }]
token_budget = 16384
max_retries = 2
# 4. Implement the plan. Writes files (jailed to the workspace). The loop target —
# max_retries here caps how many review→implement refinement rounds are allowed.
# Optional: a `writes` manifest (workspace-relative globs) hard-bounds where this
# stage may write — a FILE_WRITE outside it is blocked as scope creep. Left open
# here because the targets are task-specific; a task-scoped workflow would set e.g.
# writes = ["core/sessions/**", "testing/sessions/**"]
# Optional: `static_analysis` runs deterministic tools (compiler/detekt/formatters)
# against the patch BEFORE the reviewer (role-reliability §5). A non-clean command
# fails this stage retryably with its output fed back verbatim, so only static-clean
# code reaches the reviewer — the reviewer never spends inference on what tools catch
# for free. Commands are workspace-specific, so left commented; a Kotlin/Gradle repo
# would set e.g.
# static_analysis = ["./gradlew compileKotlin -q", "./gradlew detekt -q"]
[[stages]]
id = "implementer"
prompt = "prompts/implementer.md"
needs = ["impl_plan"]
produces = [{ name = "patch", kind = "file_written" }]
allowed_tools = ["file_read", "file_write", "file_edit", "ShellTool"]
# static_analysis = ["./gradlew compileKotlin -q", "./gradlew detekt -q"]
token_budget = 32768
max_retries = 3
# 5. Review the patch against the plan AND the analyst's acceptance criteria. The reviewer
# needs `analysis` so it judges the diff against concrete, pre-stated criteria (§5 narrow
# question) rather than whole files against taste.
[[stages]]
id = "reviewer"
prompt = "prompts/reviewer.md"
needs = ["patch", "impl_plan", "analysis"]
produces = [{ name = "review_report", kind = "review_report" }]
token_budget = 32768
max_retries = 2
# --- forward edges ---
[[transitions]]
id = "analyst-to-architect"
from = "analyst"
to = "architect"
condition_type = "artifact_validated"
condition_artifact_id = "analysis"
[[transitions]]
id = "architect-to-brief-echo"
from = "architect"
to = "brief_echo"
condition_type = "artifact_validated"
condition_artifact_id = "design"
[[transitions]]
id = "brief-echo-to-planner"
from = "brief_echo"
to = "planner"
condition_type = "artifact_validated"
condition_artifact_id = "brief_echo"
[[transitions]]
id = "planner-to-implementer"
from = "planner"
to = "implementer"
condition_type = "artifact_validated"
condition_artifact_id = "impl_plan"
[[transitions]]
id = "implementer-to-reviewer"
from = "implementer"
to = "reviewer"
condition_type = "artifact_validated"
condition_artifact_id = "patch"
# --- verdict-gated loop exit / re-entry ---
[[transitions]]
id = "review-approved"
from = "reviewer"
to = "done"
condition_type = "artifact_field_equals"
condition_artifact_id = "review_report"
condition_field = "verdict"
condition_value = "approved"
condition_operator = "eq"
[[transitions]]
id = "review-changes-requested"
from = "reviewer"
to = "implementer"
condition_type = "artifact_field_equals"
condition_artifact_id = "review_report"
condition_field = "verdict"
condition_value = "approved"
condition_operator = "neq"