okstra 0.164.0 → 0.165.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture.md +12 -8
  3. package/docs/cli.md +7 -3
  4. package/docs/for-ai/README.md +2 -2
  5. package/docs/for-ai/skills/okstra-inspect.md +2 -2
  6. package/docs/for-ai/skills/okstra-user-response.md +2 -2
  7. package/docs/project-structure-overview.md +15 -9
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/workers/antigravity-worker.md +9 -7
  11. package/runtime/agents/workers/codex-worker.md +9 -7
  12. package/runtime/agents/workers/grok-worker.md +6 -4
  13. package/runtime/agents/workers/kimi-worker.md +6 -4
  14. package/runtime/bin/okstra-antigravity-exec.sh +1 -340
  15. package/runtime/bin/okstra-claude-exec.sh +1 -178
  16. package/runtime/bin/okstra-codex-exec.sh +1 -467
  17. package/runtime/bin/okstra-provider-exec.py +165 -190
  18. package/runtime/bin/okstra-trace-cleanup.sh +14 -7
  19. package/runtime/bin/okstra-wrapper-status.py +26 -19
  20. package/runtime/prompts/lead/adapters/cmux.md +1 -1
  21. package/runtime/prompts/lead/convergence.md +36 -8
  22. package/runtime/prompts/lead/okstra-lead-contract.md +23 -1
  23. package/runtime/prompts/lead/plan-body-verification.md +9 -1
  24. package/runtime/prompts/lead/report-writer.md +1 -0
  25. package/runtime/prompts/lead/team-contract.md +3 -3
  26. package/runtime/prompts/profiles/_common-contract.md +9 -1
  27. package/runtime/prompts/profiles/_coverage-critic.md +1 -1
  28. package/runtime/prompts/profiles/_implementation-diff-review.md +3 -1
  29. package/runtime/prompts/profiles/_implementation-self-check.md +1 -1
  30. package/runtime/prompts/profiles/_implementation-verifier.md +3 -1
  31. package/runtime/prompts/profiles/implementation-planning.md +5 -3
  32. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
  33. package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
  34. package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +148 -0
  35. package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +55 -0
  36. package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +41 -0
  37. package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +44 -0
  38. package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +42 -0
  39. package/runtime/python/okstra_ctl/dispatch_core.py +5 -1
  40. package/runtime/python/okstra_ctl/dispatch_state.py +10 -0
  41. package/runtime/python/okstra_ctl/domain/provider.py +5 -1
  42. package/runtime/python/okstra_ctl/domain/worker_exec.py +102 -0
  43. package/runtime/python/okstra_ctl/domain/worker_role.py +34 -0
  44. package/runtime/python/okstra_ctl/domain/worker_stream.py +261 -0
  45. package/runtime/python/okstra_ctl/incremental_scope.py +16 -4
  46. package/runtime/python/okstra_ctl/report_html/common.py +71 -25
  47. package/runtime/python/okstra_ctl/report_html/models.py +5 -0
  48. package/runtime/python/okstra_ctl/report_html/render.py +1 -1
  49. package/runtime/python/okstra_ctl/report_html/run_usage.py +19 -0
  50. package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -0
  51. package/runtime/python/okstra_ctl/report_views.py +44 -16
  52. package/runtime/python/okstra_ctl/stage_citations.py +52 -15
  53. package/runtime/python/okstra_ctl/user_response.py +45 -29
  54. package/runtime/python/okstra_ctl/wizard.py +13 -9
  55. package/runtime/python/okstra_ctl/worker_prompt_policy.py +10 -3
  56. package/runtime/python/okstra_ctl/worker_request.py +140 -0
  57. package/runtime/python/okstra_ctl/worker_runner.py +622 -0
  58. package/runtime/python/okstra_token_usage/collect.py +8 -1
  59. package/runtime/python/okstra_token_usage/report.py +42 -0
  60. package/runtime/python/okstra_token_usage/task_totals.py +88 -0
  61. package/runtime/schemas/final-report-v1.0.schema.json +70 -0
  62. package/runtime/schemas/final-report-v2.0.schema.json +90 -0
  63. package/runtime/skills/okstra-inspect/SKILL.md +1 -2
  64. package/runtime/skills/okstra-inspect/facets/logs.md +5 -5
  65. package/runtime/skills/okstra-inspect/facets/run-audit.md +3 -3
  66. package/runtime/skills/okstra-run/SKILL.md +1 -1
  67. package/runtime/skills/okstra-user-response/SKILL.md +15 -5
  68. package/runtime/templates/report-writer-prompt-preamble.md +1 -0
  69. package/runtime/templates/reports/html/assets/base.css +8 -4
  70. package/runtime/templates/reports/html/base.template.html +12 -6
  71. package/runtime/templates/reports/html/i18n/en.json +29 -6
  72. package/runtime/templates/reports/html/i18n/ko.json +29 -6
  73. package/runtime/templates/reports/html/macros/forms.html +9 -3
  74. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +14 -19
  75. package/runtime/templates/reports/report.js +59 -26
  76. package/runtime/templates/reports/user-response.template.md +12 -8
  77. package/runtime/validators/validate-run.py +88 -7
  78. package/runtime/validators/validate_session_conformance.py +62 -1
  79. package/src/cli-registry.mjs +0 -7
  80. package/runtime/bin/okstra-wrapper-agy-stream.py +0 -61
  81. package/runtime/python/okstra_ctl/error_issue.py +0 -640
  82. package/runtime/python/okstra_ctl/issue_signals.py +0 -186
  83. package/runtime/skills/okstra-inspect/facets/error-issue.md +0 -77
  84. package/src/commands/inspect/error-issue.mjs +0 -27
@@ -123,6 +123,14 @@ DISAGREE on a `P-Var-*` item means one of:
123
123
 
124
124
  The hexagonal rule that an extracted point must declare `interfaceKind: "port"` is already machine-checked by `validators/validate-run.py` `_validate_variation_point_analysis` (it fires only for a project whose `architecture.style` is `hexagonal`). Do not re-run that mechanical check as a verdict; spend the judgement on placement and semantics instead — a point extracted as a port whose domain rule leaked into the adapter passes the validator and is still wrong.
125
125
 
126
+ `P-Opt-<N>` carries the **YAGNI judgement** and is majority-gated for the same reason as `P-Var-*`: whether an abstraction serves the stated requirement or only a forecast is a judgement about the design, not a contradiction between two spelled-out references. Raise it as `DISAGREE(e)` — an option that carries an abstraction, parameter, or configuration knob no Requirement Coverage row demands contradicts the trade-off matrix that scored it, because the complexity the matrix priced is not the complexity the option actually buys. DISAGREE on a `P-Opt-*` item under this rule means one of:
127
+
128
+ - **an abstraction nobody asked for** — a helper module, strategy / factory, indirection layer, or interface whose only justification in the plan is a caller no requirement names. A second implementation already on the table is `P-Var-*` territory and is the opposite defect: do not raise both on the same behavior;
129
+ - **a configuration knob with one value** — a flag, option object, or env-var switch that every planned call site passes identically;
130
+ - **a widened signature** — a step that adds an optional parameter no planned call site supplies.
131
+
132
+ The scope-provenance gate cannot reach these. It resolves the source of a *requirement row* and the citation of a *stage*, so unrequested work smuggled inside a legitimately-sourced stage passes it clean — the gate's own stated limit (`prompts/profiles/implementation-planning.md` "Scope provenance" → "The reach of this gate — do not over-trust it"). This verdict is that gate's missing half and the only plan-side judgement that can block on it, so a `P-Opt` item's AGREE asserts the option is free of unrequested work — not merely that it is executable. **Enforced:** `validators/validate-run.py` `_classify_plan_item_gate` — kind `e` sits in neither `_SINGLE_VOTE_BLOCKING_KINDS` nor `_ADVISORY_ONLY_KINDS`, so a `majority-disagree` on a `P-Opt-*` item blocks approval exactly like any other majority-gated defect, and one lone dissent does not.
133
+
126
134
  The semantic checks above are the plan-body enforcement layer for the schema-valid structures; `validators/validate-run.py` enforces detector coverage and references, while this worker verdict decides whether the content is implementable.
127
135
 
128
136
  Worker non-result handling (`timeout`, `error`, no result file, wrapper `cli-failure`) is identical to finding convergence: do NOT aggregate as DISAGREE, record `contract-violation`, and apply the round-level abort rule below.
@@ -364,7 +372,7 @@ verdict:
364
372
  (b) command or referenced path is not executable or is ambiguous — including an abbreviated / ellipsis / under-specified path that does not resolve as written. A command that IS declared but cannot run here because build/test dependencies are not installed is NOT (b) — answer UNVERIFIABLE,
365
373
  (c) validation signal is not observable,
366
374
  (d) rollback violates commit / dependency order — advisory only: a rollback is run by a human, so a DISAGREE(d), and any DISAGREE on a `P-Rb-*` rollback item, is recorded as dissent but never blocks approval,
367
- (e) item contradicts the trade-off matrix,
375
+ (e) item contradicts the trade-off matrix — **including a `P-Opt-*` option carrying an abstraction (helper module, strategy / factory, indirection layer, interface), a configuration knob every planned call site passes identically, or an optional parameter no planned call site supplies, when no `P-Req-*` requirement-coverage item in this same queue maps to it**: the matrix priced a complexity the option does not buy. Decide this from the queue alone — the requirement-coverage rows are in it, so this is plan-internal consistency, not a re-analysis of the brief. A behavior that already has two implementations is the opposite defect and belongs to `P-Var-*`; do not raise both on one behavior,
368
376
  (f) requirement coverage row does not map the stated requirement to a concrete satisfying option / stage / step — citing an existing option counts as concrete even if that option's paths are abbreviated (that is (b) on the option's item, not (f)).
369
377
  When you give a DISAGREE, also answer **Fixability** — `planner-fixable` if this defect can be fixed using only the code + this plan draft + the brief, `needs-user-input` if an open user clarification / external information is required.
370
378
  - **SUPPLEMENT**: The item is sound but a dependency / edge case / precondition
@@ -329,6 +329,7 @@ Every field MUST anchor its claim with at least one evidence reference — a `pa
329
329
 
330
330
  0. **Clarification Response Carried In** — render this `## 0.` heading ONLY when `{{CLARIFICATION_RESPONSE_RELATIVE_PATH}}` is non-empty. Walk every `C-*` row of the prior report's `## 1. Clarification Items` table, reconcile against new evidence, and record the outcome (`resolved` / `obsolete`) with citation before drafting the verdict. When no carry-in path was provided, OMIT the `## 0.` heading entirely — the validator fails an empty Section 0 stub. The lead calls `okstra incremental-scope` exactly once, combining answered-clarification stage impacts (`--impacted`) and changed PREP IDs (`--prep-items`); selected-option, Stage Map, or recommended-approach changes pass both CSVs empty to force full mode. Record that single decision JSON verbatim into `implementationPlanning.incrementalDecision` (`mode`, `reverifyStages`, `carryStages`, `reason`); the renderer emits the `### 0.1 Incremental Re-Verification Scope` audit block from it, and the validator fails an `incremental`-mode run whose Section 0 omits that block. In `incremental` mode this run's `planItems` MUST carry every plan-item id from the re-verified stages forward with its updated verdict; if re-verification concludes a plan item should be REMOVED, that is a signal the answer's blast radius is not local — do not drop it here, tell the lead to abandon incremental and re-route to a FULL re-verification, because the carry merge only adds prior items and would resurrect the removed item's stale verdict. After authoring the current data.json, call `okstra incremental-carry`, passing the decision's `carryStages` CSV to `--carry-stages` and its `reverifyStages` CSV to `--reverify-stages`. A `CarryError` means the stage/PREP ownership contract is unsafe: discard the partial merged output and route the run through full re-verification; never publish a partially merged report.
331
331
  1. **Clarification Items** — single unified `C-*` table; column schema (4 columns with the short fields stacked in one record-meta cell), ID convention, and rerun behaviour are owned by `_common-contract.md §Clarification request policy` (SSOT). The deprecated `5.5.9 Open Questions` / `1.1 Additional Material Request` / `1.2 User Confirmation Questions` sub-sections are removed; the validator fails reports that reintroduce them.
332
+ - **Open `Blocks=approval` rows carry `origin` and `userConfirmation`** (same SSOT). Lead's dispatch prompt MUST state, per intended blocker, which `origin` applies and what Lead did about it — the writer cannot observe either. When Lead instructed the writer to raise an item rather than decide it, that row's `origin` is `lead-directed` no matter how the workers subsequently voted on it: an instruction returning as a consensus is not a finding. Before writing such an instruction, run the confirmation sequence in [okstra-lead-contract](./okstra-lead-contract.md) "User confirmation before an approval blocker" — asking first is usually cheaper than the row.
332
333
  2. **Evidence and Detailed Analysis** — primary evidence rows (file path, line, snippet); secondary evidence / alternate interpretations. If `reference-expectations.md` lists explicit expected values, record match/gap per row.
333
334
  - **Error-analysis diagnosis and routing.** When `header.taskType` is `error-analysis`, populate the required `errorAnalysis` object. Copy `errorAnalysis.symptomVerbatim` byte-for-byte from the symptom stated in the brief's `Source Material`; do not paraphrase it. Every `causeCandidates[]` row includes the full `supportingEvidence`, `falsifyingEvidenceChecked`, `confidence`, and `disproveWith` fields. When a candidate is a step in a propagation chain rather than a competing explanation — the analysis calls it a downstream step, a second stage, or a consequence of another candidate — set its `downstreamOf` to the ids of the candidates immediately upstream of it; leave the field absent for a candidate that stands on its own. Every id listed MUST be another candidate in the same report, no row may name itself, and the links MUST NOT form a cycle; `validators/validate-run.py::_validate_cause_chain` rejects all three. This is the only place the chain is machine-readable — prose calling a candidate "the second step of the chain" while `downstreamOf` is absent leaves the report's figure claiming the candidates are alternatives. Route `errorAnalysis.routing.nextTaskType=implementation-planning` with `direction=begin-planning`, or route `errorAnalysis.routing.nextTaskType=error-analysis` with `direction=continue-investigation`; no other pairing is valid. `verdictCard.nextStep`, `finalVerdict.nextStep`, the first `recommendedNextSteps` action and command, and the unique `followUpTasks` row whose `origin` is `phase-continuation` MUST all point to the same `errorAnalysis.routing.nextTaskType` target. The schema enforces only the presence of a `phase-continuation` row. Phase validation MUST enforce exact target agreement and uniqueness through `validators/validate-run.py::_validate_error_analysis_consistency`; until that check is implemented and executed, those semantics are contract requirements rather than enforced guarantees.
334
335
  3. **Recommended Next Steps** — prioritized actions. After Phase 7's follow-up spawner runs, append a row per newly created task-key (see "Phase 6 → Phase 7 execution sequence" above). **Approval-gate consistency:** when §1 carries any `Blocks: approval` row with `Status` ∈ {open, answered}, the Verdict Card `Next Step` and the first recommended step MUST point to the clarification rerun (`resume-clarification` of the SAME task-type) — never to "flip frontmatter `approved: true` → jump straight to `implementation`". Run-prep enforces this gate (`run.py _validate_approved_plan` fail-closes on those rows and on a blocking data.json `gateResult`), so a direct-implementation next-step is an instruction the reader cannot actually follow. **Cross-project pointer rule:** for cross-project dependencies (another repo / a different top-level deployment module / a published package), `crossProjectDependencies` (§5.4 Cross-Project Dependencies) is authoritative — do NOT duplicate that substance (prerequisite work / verification signals / handoff) into `recommendedNextSteps`; put only a one-line pointer to that section (no double-recording).
@@ -177,7 +177,7 @@ After each worker subagent returns (regardless of role), Lead MUST verify the ca
177
177
 
178
178
  **Logging.** Lead records the first attempt's `cli-failure` (already emitted by the wrapper sub-agent) as-is. The retry, on success, is logged via the normal worker-completion path; on failure (second `*_RESULT_MISSING`), Lead records a single `contract-violation` entry with `--message "result-missing after 1 retry"` referencing both adapter dispatch-attempt ids and prompt-history paths.
179
179
 
180
- **Diagnostic sidecar (advisory).** Both codex/antigravity wrappers also write a heartbeat sidecar at `<prompt-path>.status.json` recording `started_ts`, `ended_ts`, `exit_code`, `duration_ms`, and the canonical `log_path` (see `scripts/okstra-wrapper-status.py` for the schema). Lead MAY read this sidecar when deciding whether the first attempt actually launched the CLI (stage=`exited`, `exit_code=0`, non-zero `duration_ms`) versus failed before reaching it (sidecar absent, or stage=`started` with no exit fields). The sidecar is best-effort — its absence is NOT by itself a reason to skip the retry; the canonical trigger remains the missing result file.
180
+ **Diagnostic sidecar (advisory).** Every CLI-worker dispatch writes a heartbeat sidecar at `<prompt-path>.status.json` recording `started_ts`, `ended_ts`, `exit_code`, `duration_ms`, and the canonical `log_path` (written by `scripts/okstra_ctl/worker_runner.py`, which every provider entrypoint shares). Lead MAY read this sidecar when deciding whether the first attempt actually launched the CLI (stage=`exited`, `exit_code=0`, non-zero `duration_ms`) versus failed before reaching it (sidecar absent, or stage=`started` with no exit fields). A run that ended abnormally after launch — an error, a Ctrl-C, or the SIGTERM/SIGHUP a pane kill or session teardown sends — closes as stage=`exited` with a `failure` string and **no** `exit_code`; read that as a failed attempt, not as a success. A SIGKILL cannot be closed by anything, so a sidecar still reading stage=`started` is not evidence that the worker is alive. The sidecar is best-effort — its absence is NOT by itself a reason to skip the retry; the canonical trigger remains the missing result file.
181
181
 
182
182
  **Rationale.** Observed failure mode: the CLI (codex/antigravity) streams its full analysis to stdout but hits its token budget or a sandbox EPERM mid-`Write` of the result file, exiting 0 with no artifact. Forwarding the partial stdout silently degrades synthesis; classifying the role as `error` without retrying gives up a recoverable signal. A single retry catches the transient class of this failure (re-dispatch with the same prompt typically succeeds when the underlying cause was an intermittent sandbox lock or a token-budget spike) while bounding the retry cost to a known upper bound (~2× the original wrapper budget per role). The reading-plan exception in item 5 covers the class this rationale does not: a worker that timed out because its prompt asked it to read too much fails identically on an identical retry, so the second attempt must change the reading plan or it is knowingly wasted.
183
183
 
@@ -241,13 +241,13 @@ omits either header, the worker MUST return `<WORKER>_ERRORS_PATH_MISSING`
241
241
  without proceeding.
242
242
 
243
243
  - `cli-failure` events are recorded by the wrapper subagent itself (Codex / Antigravity), but **directly to the run-level error log** via `okstra error-log append-observed --error-type cli-failure ...` — NOT via the sidecar. The sidecar is an in-process tool-failure channel only.
244
- - **Wrapper invocation arity.** Both `okstra-codex-exec.sh` and `okstra-antigravity-exec.sh` take three required positional arguments plus three optional ones, matching their usage line verbatim: `<project-root> <model-execution-value> <prompt-path> [worktree-path] [role] [idle-timeout-seconds]`. The fourth (worktree) argument is **mandatory for implementation phase** and optional otherwise — when passed it must be an existing directory (the wrapper hard-fails otherwise). For codex it becomes `--add-dir <worktree>` (sandbox write access); for antigravity it is appended to `--include-directories`. Omitting it during implementation causes the codex sandbox to reject every Edit/Write targeting the worktree with EPERM. Workers extract the path from the `**Worktree:**` / `EXECUTOR_WORKTREE_PATH` / `cwd for every mutating command:` line in the lead prompt. The optional fifth `<role>` is folded into both the caller (worker) pane title `<cli>-<role>` and the sibling trace-pane title `<cli>-<role>-tail` (e.g. `codex-executor` ↔ `codex-executor-tail`; antigravity uses the `agy-` prefix). The selected adapter maps the functional assignment identity to this value; when the value is absent, pass the literal `worker` (the wrapper also defaults to `worker` if the argument is omitted). The optional sixth `<idle-timeout-seconds>` overrides the wrapper's stdout-idle watchdog (default 600s, or 1500s when `<role>` is `executor` / `verifier` — those roles run silent build+test suites — see "No external timeout on wrapper subagents" below).
244
+ - **Wrapper invocation arity.** Every `okstra-<provider>-exec.sh` entrypoint takes the same three required positional arguments plus three optional ones: `<project-root> <model-execution-value> <prompt-path> [worktree-path] [role] [idle-timeout-seconds]`, optionally followed by the flag `--presentation live|quiet`. The fourth (worktree) argument is **mandatory for implementation phase** and optional otherwise — when passed it must be an existing directory (preflight hard-fails otherwise, exit 68). It is added to the worker's write scope, which every provider receives as repeated `--add-dir` (codex names the project root with `-C` instead of repeating it). Omitting it during implementation causes the codex sandbox to reject every Edit/Write targeting the worktree with EPERM. Workers extract the path from the `**Worktree:**` / `EXECUTOR_WORKTREE_PATH` / `cwd for every mutating command:` line in the lead prompt. The optional fifth `<role>` names the dispatched role (`executor`, `verifier`, `worker-reverify-r1`, …); the selected adapter maps the functional assignment identity to this value, and when it is absent, pass the literal `worker`. The optional sixth `<idle-timeout-seconds>` overrides the role's idle budget (default 600s, or 1500s when `<role>` is `executor` / `verifier` — those roles run silent build+test suites — see "No external timeout on wrapper subagents" below). `--presentation` names the surface the progress has to land on. `live` renders it into the worker's own pane and is passed only by a backend that opened one; `quiet` withholds it and prints only the closing text, and is what a `cli-wrapper` subagent dispatch passes because its stdout is the calling agent's context window. Omitting the flag means `quiet` — the flag is never a way to ask for a pane that does not exist. Do not drop it when relaying a command line.
245
245
  - **Background dispatch + polling contract (Codex / Antigravity wrappers).** Both wrapper subagents MUST start their CLI through the selected adapter's asynchronous execution mapping and await the same handle until it reports terminal completion, capped at 30 minutes (1800s) of wall-clock elapsed time. The adapter's await operation is the wait primitive; do not add a standalone sleep or build shorter-sleep loops to bypass a host constraint. This rule applies in **every phase**. Recording responsibilities:
246
246
  - Successful completion: return the wrapper's accumulated stdout from the terminal await result. No log entry.
247
247
  - Non-zero `exit_code`: record a `cli-failure` to the run-level error log with the real `exit_code` and observed `duration-ms`.
248
248
  - Polling cap reached: perform a one-shot **mtime-grace check** on the wrapper's live log (`<prompt>.log`). If the log was written within the last 90 seconds and grace has not yet been applied, extend the cap from 1800s to 2100s and continue awaiting. Otherwise call the selected adapter's termination mapping, record `cli-failure` with `--exit-code 124 --duration-ms <observed_ms> --message "<wrapper> exceeded polling cap (grace=<applied|not-applied>, last_mtime_age=<n>s)"`, then return the language-specific `*_CLI_TIMEOUT` sentinel.
249
249
  - The selected adapter owns runtime-session accounting for the full wrapper window; core retains only the observed start/end event boundaries.
250
- - **No external timeout on wrapper subagents.** The codex/antigravity wrapper owns BOTH of its timeout mechanisms, and they are the only two: (1) the subagent's polling cap (30min + optional 5min mtime grace), and (2) the wrapper script's internal stdout-idle watchdog — if the CLI writes nothing to its log for `<idle-timeout-seconds>` (6th positional arg, default 600s, or 1500s for the build-running `executor` / `verifier` roles), the script TERM/KILLs the CLI and records a `timeout` stage in the status sidecar (`okstra-codex-exec.sh` / `okstra-antigravity-exec.sh`). A long-running but *chatty* CLI is never idle-killed; a silent hang is reaped at the idle cap instead of burning the full 30 minutes. Lead MUST NOT impose a separate `dispatch_worker` timeout, an outer `Bash` wall-clock deadline, or any other mechanism that terminates the subagent before the wrapper's own caps fire. Doing so reproduces the historical failure mode that motivated this rule: Lead aborts the subagent at e.g. 18 minutes, the subagent returns nothing, and Lead classifies the role as "no response" while the underlying CLI was actively working. The caps are calibrated so that, combined with Lead's redispatch policy (see "Lead Redispatch Policy on Result-Missing"), a recoverable single-run failure costs at most ~70 minutes of wall-clock — predictable enough to plan around. If a specific run requires a tighter cap, pass a lower `<idle-timeout-seconds>` or lower the polling cap in the wrapper subagent's polling contract (single source of truth), NOT by layering Lead-side timeouts.
250
+ - **No external timeout on wrapper subagents.** The wrapper dispatch owns BOTH of its timeout mechanisms, and they are the only two: (1) the subagent's polling cap (30min + optional 5min mtime grace), and (2) the shared runner's stream-idle watchdog — if the CLI produces nothing on either stream for `<idle-timeout-seconds>` (6th positional arg, default 600s, or 1500s for the build-running `executor` / `verifier` roles), the runner TERM/KILLs the CLI's whole process group and marks the status sidecar timed out (`scripts/okstra_ctl/worker_runner.py`). Idle is measured from stream arrival, not from the log file's mtime. A long-running but *chatty* CLI is never idle-killed; a silent hang is reaped at the idle cap instead of burning the full 30 minutes. Lead MUST NOT impose a separate `dispatch_worker` timeout, an outer `Bash` wall-clock deadline, or any other mechanism that terminates the subagent before the wrapper's own caps fire. Doing so reproduces the historical failure mode that motivated this rule: Lead aborts the subagent at e.g. 18 minutes, the subagent returns nothing, and Lead classifies the role as "no response" while the underlying CLI was actively working. The caps are calibrated so that, combined with Lead's redispatch policy (see "Lead Redispatch Policy on Result-Missing"), a recoverable single-run failure costs at most ~70 minutes of wall-clock — predictable enough to plan around. If a specific run requires a tighter cap, pass a lower `<idle-timeout-seconds>` or lower the polling cap in the wrapper subagent's polling contract (single source of truth), NOT by layering Lead-side timeouts.
251
251
  - `contract-violation` events (C) are recorded by Lead via `okstra error-log append-observed --error-type contract-violation ...` after inspecting worker outputs.
252
252
  - Lead's responsibility regarding the sidecar is to dump it to the run-level error log via `okstra error-log append-from-worker` after each worker terminates; Lead does not write into the sidecar.
253
253
 
@@ -11,7 +11,7 @@ profile document.
11
11
  - **Phase 4 / 5 (independent analysis)**: every analyser in the resolved provider assignment roster produces findings independently and has no access to another worker's output. `report-writer` does not analyse.
12
12
  - **Phase 5.5 (convergence — peer review by workers)**: workers peer-review each other's findings across up to `effectiveMaxRounds` rounds; the lead mediates but does not vote. See `prompts/lead/convergence.md` for the round protocol (replay of findings, `AGREE` / `DISAGREE` / `SUPPLEMENT` verdicts), queue invariants, and final classification (`full-consensus` / `partial-consensus` / `contested` / `worker-unique`). For `requirements-discovery`, `error-analysis`, `implementation-planning`, `project-analysis`, `feature-analysis`, and `change-impact-analysis` this phase runs in **adversarial mode** (`convergence.adversarial=true`): verifiers try to refute each finding against its cited evidence and the burden of proof sits on the claim — see that skill's §"Adversarial Verification Mode".
13
13
  - Do NOT conclude "no peer review happens" from the roster alone — every profile that lists ≥2 analyser workers runs convergence by default (`convergence.enabled=true` in `task-manifest.json`).
14
- - **provider-unavailable fallback (tolerance).** A worker dispatch can fail to produce a result for two distinct reasons, and both take the same recovery path. (1) **Pane budget:** the dispatch is rejected with `no room for another tmux split` (or an equivalent teammate-pane creation failure). (2) **Sandbox CLI-start failure (non-tmux path):** an external CLI worker wrapper exits non-zero within seconds with empty stdout and its live-log shows `operation not permitted`. In either case the lead spends the one shared retry budget through the assignment's recorded runner. If the provider is still unavailable, record that terminal status and continue only under the convergence quorum rules; never replace it silently with a fixed provider or count a substitute as the original provider's vote. Completed external-CLI trace panes are reclaimed by the selected runtime adapter's resource lifecycle. (This is a prompt instruction, not a code-enforced gate.)
14
+ - **provider-unavailable fallback (tolerance).** A worker dispatch can fail to produce a result for two distinct reasons, and both take the same recovery path. (1) **Pane budget:** the dispatch is rejected with `no room for another tmux split` (or an equivalent teammate-pane creation failure). (2) **Sandbox CLI-start failure (non-tmux path):** an external CLI worker wrapper exits non-zero within seconds with empty stdout and its live-log shows `operation not permitted`. In either case the lead spends the one shared retry budget through the assignment's recorded runner. If the provider is still unavailable, record that terminal status and continue only under the convergence quorum rules; never replace it silently with a fixed provider or count a substitute as the original provider's vote. Completed external-CLI worker panes are reclaimed by the selected runtime adapter's resource lifecycle. (This is a prompt instruction, not a code-enforced gate.)
15
15
  - Dual-audience final-report contract (shared):
16
16
  - data.json is the sole authored report artifact. AI handoff Markdown and human HTML are independently derived from it; neither derived artifact is the other's source.
17
17
  - User-facing information belongs in `humanSummary` and the selected task block's `userNarrative`; it must not exist only in Markdown. The HTML human main body explains the result with those fields plus task facts.
@@ -57,9 +57,17 @@ profile document.
57
57
  - `adr-candidate:` → handled by `implementation-planning`; carry forward without modification. Approved decision files land only at `<PROJECT_ROOT>/.okstra/decisions/<NNNN>-<slug>.md`.
58
58
  - `general:` → free-form; classify per the standard `Clarification Items` rules.
59
59
  - Any decision in this run that contradicts the brief's `Source Material` must be raised back to the reporter via a `Clarification Items` row; it must NOT be silently overridden. Disagreement with the reporter is allowed only after the row is resolved.
60
+ - **User instruction outranks the material it points at (BLOCKING).** The rule above governs *this run's own* decisions against the source material. It does not govern the user's: when the user's run directive, brief, or in-session instruction names a document and says to apply it, that instruction outranks any reservation written *inside* that document — `user decision required`, `do not adopt standalone`, `needs sign-off`, and their equivalents. Such a note records what the document's author (usually an earlier agent) did not have authority to settle; the user naming the document **is** that authority arriving. Reading the note as still-open re-asks a question the user just answered, and every item the instruction covers gets deferred by the one line meant to protect it.
61
+ - An instruction that points at a document covers **every** item in it. Do not carve out the items the document flagged as needing a decision and leave the rest applied — that is the split that turns one instruction into a blocker.
62
+ - When the instruction genuinely does not reach an item — the document names a choice the user's words do not cover — do NOT write a `Clarification Items` row for it. Ask the user directly at that moment, per [okstra-lead-contract](../lead/okstra-lead-contract.md) "User confirmation before an approval blocker". The row is the outlet of last resort, after asking has failed or was impossible.
63
+ - **Enforced:** `validators/validate-run.py` `_validate_open_approval_blocker_provenance` requires `origin` and `userConfirmation` on every open approval blocker and rejects a lead-authored one raised with nobody to ask; `validators/validate_session_conformance.py` `_check_user_confirm_checkpoints` requires the matching `PROGRESS: user-confirm <C-NNN>` line for every row the report claims the user was asked about.
60
64
  - This contract is the single authority on brief consumption. Phase-specific addenda may *tighten* these rules but may not relax them.
61
65
  - Clarification request policy (shared — applies whenever a profile uses `## 1. Clarification Items`):
62
66
  - Schema-v2 final reports author `clarificationItems[]` in data.json; task-specific HTML renders the question and response controls directly from those IDs, and AI handoff Markdown renders the same array as one headed section per row for the next agent. The remaining table-layout rules describe schema-v1 compatibility and analysis-worker result tables only.
67
+ - **Every row that is still `open` and carries `Blocks=approval` records two more fields.** Withholding approval is the most expensive thing a report does to a run, and until these fields existed a blocker could not be told apart from a question nobody had put to the user.
68
+ - `origin` — who raised it. `worker-finding` (an analyser or verifier reached it on its own evidence), `material-gap` (neither the brief nor the codebase answers it), or `lead-directed` (the lead's own judgment, **including anything the lead instructed a worker to raise**). A lead that seeds its conclusion into a worker prompt and then reports the worker's agreement as an independent finding has mislabelled the row; that shape is what let one run block on a question its own lead had authored.
69
+ - `userConfirmation` — what happened before the row was written. `asked-and-answered`, `asked-awaiting` (asked, no answer yet), or `deferred-no-interactive-session` (this run had no user to ask). An answered question stops being a blocker: record the answer in `userInput`, move `status` to `answered`, and let the plan proceed.
70
+ - Neither field is required once `status` is `answered` / `resolved` — the record lives in `userInput` by then.
63
71
  - **Legacy canonical column schema (must match `templates/reports/final-report.template.md` §1 exactly):** every `## 1. Clarification Items` table has exactly these 4 columns, in this order:
64
72
  `| <record-meta> | Statement | Expected form | User input |` (the first header is the i18n `columns.recordMeta` label — `Record`).
65
73
  The five short fields (ID, Ticket ID, Kind, Blocks, Status) are stacked inside the single record-meta cell, one per line separated by `<br>`, in this fixed order (mirrors the §2.1 Primary-Evidence meta column):
@@ -14,4 +14,4 @@ mode:" when the include directive (placed at column 0) is resolved in-place.
14
14
  Do NOT write the literal include directive token in this file's body — the
15
15
  resolver matches it anywhere and would recurse on this file itself.
16
16
  -->
17
- - **Coverage critic (opt-in)**: when `convergence.critic.enabled=true` (chosen via the okstra-run picker or `--critic`), a reused-worker critic pass is dispatched concurrently with the first convergence reverify round to surface missed findings; its gaps are merged only after a 1-round adversarial reverify that follows convergence. See `prompts/lead/convergence.md` "Coverage critic pass".
17
+ - **Coverage critic (opt-in)**: when `convergence.critic.enabled=true` (chosen via the okstra-run picker or `--critic`), a reused-worker critic pass is dispatched concurrently with the first convergence reverify round to surface **both** findings nobody covered and work the findings propose that no requirement asked for (`category: "unrequested-scope"`); its candidates are judged only after a 1-round adversarial reverify that follows convergence. The two halves are disposed of differently — a contested coverage gap is dropped as a hallucination, while a contested over-scope candidate is recorded as a `## 5. Missing Information and Risks` row instead of vanishing. See `prompts/lead/convergence.md` "Coverage critic pass".
@@ -27,7 +27,9 @@ Do not scan holistically and stop when it "looks fine". Work the matrix exhausti
27
27
 
28
28
  1. **List every changed file:** `git diff --name-only <stage-base>..HEAD`. That list is your worklist — cover it to the end; no sampling.
29
29
  2. **Classify each file and select its rule set** from the conventions the preflight already loaded (do NOT re-derive the rules here — read them from the routed pack):
30
- - any source file → `clean-code.md`: truthful + standalone names, one identifier one meaning per file (a key parameter must not take the name a sibling signature gives the entity), plain-English summary test, single-purpose ≤50-line functions, DRY (incl. scattered domain literals + documented forks), no magic numbers, shallow nesting, comments explain why.
30
+ - any source file → `clean-code.md`: truthful + standalone names, one identifier one meaning per file (a key parameter must not take the name a sibling signature gives the entity), plain-English summary test, single-purpose ≤50-line functions, DRY (incl. scattered domain literals + documented forks), YAGNI, no magic numbers, shallow nesting, comments explain why.
31
+ - The YAGNI cell also counts callers, not just their existence: an abstraction layer this diff *introduces* — helper module, strategy / factory / builder, indirection or wrapper layer, interface or abstract base — with exactly **one** caller after the change is an inline candidate under `overview.md` core principle 2 ("name the second caller now, or inline"). Collapse it into that call site unless one of four exits applies and you record which: the approved plan declares it as a test seam or variation-point extraction, the project declares `architecture.style = hexagonal` and it is a port at the domain boundary, it breaks a real import cycle, or its single caller is a published package's public API. A function extracted only to get a body under the 50-line cap is not this finding.
32
+ - The YAGNI cell is a grep, not an impression: for every identifier this diff adds — and every one whose last caller it removes — search the whole repository and name the caller you found. A declaration-only, test-only, or commented-out hit is not a caller. Zero callers means delete it here (or fold an added parameter back into its single call site), unless the approved plan reserves it for a named later stage or something outside project code calls it (framework entrypoint, implemented interface method, migration hook, published-package API) — in which case record that justification in the audit note. Fixing it now is cheaper than the verifier's Static design gate, where the same finding is blocking and fails the stage (`_implementation-verifier.md` "Caller-less identifier").
31
33
  - a file that wraps a third-party / library call → `clean-code.md` "Wrapping a third-party call": the wrapper adds behaviour the library does not already provide (read the installed library source before keeping a recovery branch — a `catch` repeating the library's own retry recovers nothing), no comment names a condition the call site does not establish, and a rethrow keeps the original error as `cause`.
32
34
  - a file that decides, mutates, or persists state → `clean-code.md` "Mutation and state boundaries": decide on the direct identifier rather than a status/flag proxy, capture before-state in one snapshot ahead of the mutating boundary, update only this work's owned fields on an existing row, re-read state before calling a zero-affected-rows write success or failure, put priority-between-inputs in a named domain function, and keep error messages to what was actually observed. Also check that no state union/enum was re-declared beside an authoritative one the domain or a dependency exports.
33
35
  - test file (`*.spec.*` / `*.test.*` / `test_*.py` / `*_test.go` …) → `clean-code.md` "Testing discipline": no self-mocking of the SUT, behavioral (outcome) assertions not interaction-only, no tautological delegation assertion, no effect claimed under its own mock, shared-fixture defaults left on the ordinary path, setup values that actually separate the scenarios, every new test helper/mock used by a test in this same diff, no positional mock-argument access (`rg 'mock\.calls'`), every branch this diff adds covered by a test that fails when the branch body is deleted, assertions on the last write to a record rather than an intermediate one, and test titles naming their unit plus the single condition each case isolates.
@@ -22,7 +22,7 @@ fails, fix it or surface the violation — do not claim done on a failing item.
22
22
  - verification: ran the actual build/test — paste the exact command line and its real exit code / output tail, never a paraphrased "tests pass"
23
23
  - acceptance: each implemented plan step's `Acceptance:` condition is observably met — cite the RED→GREEN test (or the command output) proving it per step, not one global "acceptance met"; and each `Test case (success|boundary|failure)` the plan declared for this stage has a matching test in the diff (name the test), so declared edge/failure cases are not silently uncovered
24
24
  - mutation check: enumerate every branch this diff adds (`catch`, guard, early return, `else`), plus at least one test it added or changed. For each entry, break what it covers on purpose — delete the branch body, return the wrong value, skip the write — confirm a test FAILS, then restore; cite the branch (or test) and the observed failure per entry, not one global "mutation verified". A branch whose deletion leaves the suite green is untested, not covered — an untested recovery path is indistinguishable from one that cannot fire. "Verified" claimed without this check plus the real command output above is not a verified claim
25
- - cleanup: no dead or commented-out code left behind
25
+ - cleanup: no dead or commented-out code left behind, and no caller-less identifier — list every identifier this diff added or whose last caller it removed, and cite the caller you grepped for each one (or the plan step / non-project-code contract point that justifies keeping it with none). "No new identifiers" is a valid answer; a global "cleaned up" is not
26
26
 
27
27
  Close with a `Self-check coverage:` line naming the files you verified the
28
28
  per-file items against, so the Coverage footer above and this gate reconcile.
@@ -176,7 +176,9 @@ Re-running commands proves the diff *builds and passes*; it does NOT prove the d
176
176
  - **Positional mock-argument access:** `mock.calls[<n>][<m>]` (or the framework equivalent) used to inspect arguments instead of an intent-revealing `toHaveBeenCalledWith` / explicit-absence assertion. Detect with `rg 'mock\.calls'` over the changed test files.
177
177
  - **Non-separating test data:** two scenarios whose setup values and assertions are identical, so a wrong implementation passes both; or new test tooling added in this diff (mock, state setter, repository branch) that no test calls.
178
178
  - **Effect claimed under its own mock:** a test presented as covering an effect whose producing path is replaced by a mock inside that same test. The mock's presence in the harness is not evidence the branch behind it works.
179
- - **Advisory findings (recorded as recommendations; verdict MAY still PASS):** function >50 effective lines, a single body mixing read+write stages, weak readability, a missing-but-non-critical outcome assertion, newly orphaned private/public code that is safe to remove but not on a critical path, weak-but-not-misleading names, priority-between-inputs policy inlined in a service condition instead of a named domain function, an error message asserting a cause the code never observed, a memory / concurrency / batching change with no test pinning the bound it claims, or (hexagonal overlay only) a service dependency this diff adds or modifies that injects a concrete adapter instead of a port — record it with the port sketch; advisory only while the project has not declared `architecture.style = hexagonal`, since that declaration — and only that one — promotes exactly this item to blocking per the declared-style bullet above, while a declared `layered` binds dependency direction instead and leaves this item advisory; an existing convention of concrete injections does not convert this one to `clean`, it is the debt the rule pays down. These land in the verifier result as `should-fix` / `nit` recommendations, not as a `FAIL`.
179
+ - **Single-caller abstraction (KISS):** an abstraction layer this diff introduces — a helper module, a strategy / factory / builder, an indirection or wrapper layer, an interface or abstract base — that has exactly one caller after the change, where inlining it at that call site would be simpler. The rule is `overview.md` core principle 2: name the second caller now, or inline. Four exits are legitimate and each must **cite its evidence**: the approved plan declares it (a `testSeams[]` boundary or a variation-point extraction — cite the plan step); the project declares `architecture.style = hexagonal` and this is a port at the domain boundary (a port with one adapter is the normal shape, not a violation); it breaks a real import cycle (name both modules); or its single caller is the public API of a published package. Absent one of those, cite `path:line` and recommend the inlined shape. Do NOT raise this on a function extracted purely to bring a body under the 50-line cap — that is principle 5 doing its job — nor on a pre-existing abstraction this diff merely edits.
180
+ - **Caller-less identifier (YAGNI / orphan):** an identifier this diff leaves with zero callers — either newly added and never called (a speculative parameter, an optional config object, a "future-proof" hook, an exported helper whose only caller is hypothetical), or pre-existing code orphaned because this diff removed its last caller. Grep the identifier across the whole repository rather than the diff alone; a hit inside its own declaration, its own test, or a commented-out line is not a caller. Two exits are legitimate and must be **stated with their evidence**, never assumed: an identifier the approved plan reserves for a named later stage (cite the plan step), and a contract point something other than project code calls (a framework entrypoint, an implemented interface method, a migration hook, the public API of a published package — cite the caller or the registration). Anything else is work no requirement asked for: recommend deleting it, or folding an added parameter back into its single call site. A caller-less identifier is the one YAGNI violation that survives a green pipeline unchanged, which is why it grades here rather than as a recommendation.
181
+ - **Advisory findings (recorded as recommendations; verdict MAY still PASS):** function >50 effective lines, a single body mixing read+write stages, weak readability, a missing-but-non-critical outcome assertion, weak-but-not-misleading names, priority-between-inputs policy inlined in a service condition instead of a named domain function, an error message asserting a cause the code never observed, a memory / concurrency / batching change with no test pinning the bound it claims, or (hexagonal overlay only) a service dependency this diff adds or modifies that injects a concrete adapter instead of a port — record it with the port sketch; advisory only while the project has not declared `architecture.style = hexagonal`, since that declaration — and only that one — promotes exactly this item to blocking per the declared-style bullet above, while a declared `layered` binds dependency direction instead and leaves this item advisory; an existing convention of concrete injections does not convert this one to `clean`, it is the debt the rule pays down. These land in the verifier result as `should-fix` / `nit` recommendations, not as a `FAIL`.
180
182
  - **Output.** Every finding — blocking or advisory — is a structured item in the verifier's worker result (`path:line`, rule, severity, suggested fix) so it carries into Phase 5.5 convergence and the final report. A blocking hit sets the verifier verdict to `FAIL` with the rule cited, using the same verdict machinery as the Discrepancy rule above. The Okstra lead MUST NOT silently downgrade a cited blocking finding to advisory during synthesis; an override requires a concrete cited reason, exactly as for the Discrepancy rule.
181
183
 
182
184
  ### Fix-run incremental scope (applies when the profile carries a "Fix-Run Carry" block)
@@ -26,6 +26,7 @@
26
26
  - skim recent commits touching those files (`git log -- <path>`) to surface in-flight work or contested areas
27
27
  - **sibling exploration (variation-point evidence)**: read the `Related Task Graph` sibling / `related-to` tasks' done artifacts *and the code they actually landed* — a done report is a claim, the diff is the fact. When a sibling already implements the same behavior for another resource, that is a variation point: register the existing implementation in `variationPointAnalysis.evidence` and weigh an option that refactors it behind a shared interface instead of adding a second parallel implementation alongside it. Absent an explicit graph edge, still surface a same-behavior implementation you saw in the files or `git log` output already inspected above — an unrecorded edge does not make the duplication less real.
28
28
  - **codebase-first ambiguity resolution**: any ambiguity that can be answered by `Read` / `Grep` MUST be resolved that way and recorded with file:line evidence. Only ambiguities that genuinely require a human decision are escalated as `Clarification Items` rows. Writing a clarification row for something the code already answers is a defect of this phase.
29
+ - **directive-first ambiguity resolution** (the same rule, pointed at the user instead of the code): any ambiguity the run's directive, the brief, the carried-in `user-responses/` sidecars, or the user's in-session instruction already answers MUST be resolved that way and recorded with the quoted instruction. Writing a clarification row for something the user already decided is the same defect as writing one for something the code already answers — and it costs more, because the row withholds approval until a whole separate answer cycle closes it. When an instruction points at a document, treat every item in that document as decided, including the ones the document itself flagged as needing a decision (shared rule: `_common-contract.md` "User instruction outranks the material it points at").
29
30
  - flag any requirement that is ambiguous, contradictory, or missing success criteria — register each one as a row in the report's `## 1. Clarification Items` table with `Blocks=approval` instead of guessing
30
31
  - read `<PROJECT_ROOT>/.okstra/glossary.md` and `<PROJECT_ROOT>/.okstra/decisions/` titles if present. Absent okstra memory files are the normal state — do not error. Treat the brief's `terminology:*` resolutions from `requirements-discovery` (if any) as authoritative; if missing, resolve any remaining fuzzy term as a `Blocks=approval` clarification row.
31
32
  - **spec-settled short-circuit**: when the brief already carries a decision-complete design the reporter has confirmed (a `requirements-discovery` outcome whose approach sits behind a `[CONFIRMED …]` marker, or a verbatim reporter-approved plan in `Source Material`), do NOT re-litigate the settled decision. Present that design as the `Recommended Option`, and record the alternatives it already weighed as the remaining Option Candidate(s) tagged `(considered & rejected upstream — <why>)` rather than manufacturing fresh competing options to fill the slot. The Stage Map, validation, gates, and the ≥2-candidate shape all still apply — this short-circuits re-deliberation, not the plan's structure.
@@ -41,7 +42,7 @@
41
42
  - **Files that change together live together**: split by responsibility, not by technical layer. Penalize options that scatter one logical change across unrelated layers.
42
43
  - **Follow established patterns**: in existing codebases, conform to current conventions. Targeted cleanup of a file you are already modifying is acceptable; unrelated refactors are not.
43
44
  - **Variation-point extraction (OCP)**: when the same behavior is served by two or more resources / implementations — stated in the brief, or foreseeable from a sibling task or the code you inspected — the plan MUST record it in `variationPointAnalysis` and include an option that extracts the variation point behind an interface (a port, or a strategy the next implementation plugs into), scored against the non-extracted option in the trade-off matrix. Penalize an option that branches on resource identity inside a service (one `if` / `switch` arm per implementation): adding the next implementation then means editing that same call site again, which is the closed-for-extension shape this principle exists to catch. This does not contradict YAGNI below: YAGNI drops *speculative* variation (a second implementation nobody named), while a behavior with two implementations already on the table is a present fact, not a forecast. **Enforced:** the `variationPointAnalysis` bullet under `Required deliverable shape` names the schema / validator / `P-Var-*` enforcement points.
44
- - **YAGNI ruthlessly**: drop features, abstractions, and configuration knobs that do not serve the stated requirement.
45
+ - **YAGNI ruthlessly**: drop features, abstractions, and configuration knobs that do not serve the stated requirement. The test is a *present* caller, not a plausible one — an abstraction whose only justification is a requirement nobody has stated is this rule's target, while a behavior with two implementations already on the table belongs to `Variation-point extraction` above. **Enforced:** the §5.5.9 plan-body verification round raises it as a `P-Opt-*` `DISAGREE(e)`, majority-gated (`prompts/lead/plan-body-verification.md` "`P-Opt-<N>` carries the **YAGNI judgement**"), and one phase later the `implementation` verifier's Static design gate fails the stage on a caller-less identifier (`prompts/profiles/_implementation-verifier.md` "Caller-less identifier (YAGNI / orphan)"), which the executor's `Pre-commit diff review sweep` is expected to have already removed. Note what is NOT enforced: no validator reads plan prose for a speculative abstraction, so passing `Scope provenance` below is not evidence this rule was applied — the verdict is a worker judgement or it is nothing.
45
46
  - **Project review-rule preflight**: apply a project review rule pack only when the task brief's `Source Material` or `Reporter Confirmations` cites its exact `SKILL.md` path. Read only that cited file and the `references/*.md` files it directly names. Do not search parent directories or host skill catalogs. Do not run the PR-review workflow here; extract only the rules. For Fonts Ninja-style TS/NestJS review packs, this means planning away known review findings before code exists: shared transforms instead of duplicate helper stacks, behavioral tests instead of collaborator-tautology assertions, domain rules in domain modules rather than repositories/adapters, domain objects under `domain/`, plain-English functions, truthful/specific names, and no dead APIs introduced by the plan.
46
47
  - Expected output emphasis:
47
48
  - feasible plan options
@@ -159,8 +160,9 @@
159
160
  - `contract:<rule>` — an artifact okstra's own phase contract mandates, so it has no brief line to cite. The allowlist is exactly `decision-record-step` (the §5.4 Decision Drafts materialization step) and `glossary-step` (the glossary proposal step); the SSOT is `scripts/okstra_ctl/scope_provenance.py`. Never widen this form to launder work the brief did not ask for.
160
161
  An item you can give none of these three sources to is **not a requirement and not a stage**. Its only admissible outlet is a `## 1. Clarification Items` row with `Blocks=approval`, carrying the recommendation format from `_clarification-recommendation.md`. Do not fold it into an option, a stage, or a step "while we are in here" — that is the scope expansion this rule exists to stop. This makes concrete the planning-input rule that any change beyond what `Requirement Summary` explicitly demands is out of scope by default. **Enforced:** `validators/validate-run.py` `_validate_requirement_provenance` (source resolution) and `_validate_stage_has_requirement` (no stage without a requirement).
161
162
  - **The reach of this gate — do not over-trust it.** What is mechanically enforced is the *form* of each source, that a cited `brief:` id is one the brief actually declares (or, on a pre-end-state brief, that the heading literally exists), that a `derived:` chain terminates without cycling, and that no stage is uncited. What is **not** enforced is whether the cited source genuinely demands the requirement. The id form closes the older loophole — a brief no longer offers generic headings any invented work could be hung on — but it leaves one open: attaching a requirement the cited `EB-NNN` does not actually ask for still parses clean, because no machine reads that id's sentence and compares it to your row. The gate's value is that it forces every item to name a specific reporter line and makes fabrication explicit and auditable — judging whether that line actually demands the item remains a reviewer / `DISAGREE(f)` responsibility, and passing this gate is never evidence that the scope is justified.
162
- - **Stage citation format:** the reverse check reads each `Covered by` cell as prose, so a stage counts as cited only when its number is anchored to a `Stage` / `Stages` word on the same line. These all read: `Stage 2`; `Stage 1, Stage 2, Stage 3`; `Stages 1, 2, 3`; `Stages 1, 2, and 3`; ranges (`Stages 1-3`, `Stages 1 to 3`, `Stages 1 through 3`); and `and` / `&` conjunctions. A bare number with no `stage` word anchoring it is NOT read as a citation, so `covered by the recommended option, step 4` cites nothing. **Enforced:** `validators/validate-run.py` `_validate_stage_has_requirement` fails the plan when any Stage Map stage is cited by no coverage row.
163
- - Because that reader only sees prose, it cannot tell a citation from a mention: `Stage 1 (superseded by Stage 2)` still counts Stage 1 as cited, and one row citing `Stages 1-64` rubber-stamps every stage in the map. Cite the stages a requirement is actually satisfied by — not stages merely mentioned, and never a blanket range standing in for the work of checking.
163
+ - **Stage citation format — enumerate, never range (scale gate):** the reverse check reads each `Covered by` cell as prose, so a stage counts as cited only when its number is anchored to a `Stage` / `Stages` word on the same line. These read: `Stage 2`; `Stage 1, Stage 2, Stage 3`; `Stages 1, 2, 3`; `Stages 1, 2, and 3`; and `and` / `&` conjunctions. A bare number with no `stage` word anchoring it is NOT read as a citation, so `covered by the recommended option, step 4` cites nothing. **A range cites only its two endpoints:** `Stages 1-3` cites 1 and 3, and stage 2 stays uncited — write every stage out. Range syntax (`-`, `to`, `through`) is still parsed, so `Stages 7-8` is a valid two-stage citation; what it cannot do is stand in for an interior nobody named. **Enforced:** `validators/validate-run.py` `_validate_stage_has_requirement` via `okstra_ctl.stage_citations.enumerated_stage_numbers` fails the plan when any Stage Map stage is cited by no coverage row.
164
+ - **Why enumeration is the scale gate.** The number of stages a plan carries is not bounded by any threshold — a genuinely large requirement may need many, and okstra does not guess a ratio. What IS bounded is how cheaply a plan can *claim* coverage of them: one `Stages 1-64` cell used to satisfy the reverse check for the whole map while the planner confirmed nothing, so scale grew for free. Enumeration prices it — every stage you claim costs you the act of naming it and asking whether this requirement is really satisfied there. A plan that cannot bring itself to type the numbers is telling you the stages are not all needed. The typing is the confirmation, so do not batch it mechanically: a row listing `Stages 1, 2, 3, ..., 12` you did not check one by one is the same rubber stamp with more characters.
165
+ - Because that reader only sees prose, it still cannot tell a citation from a mention: `Stage 1 (superseded by Stage 2)` counts Stage 1 as cited. Cite the stages a requirement is actually satisfied by, not stages merely mentioned.
164
166
  - **Requirement Coverage (mandatory, §5.5.8):** one row per concrete requirement from the task brief / packet. Assign stable IDs `R-001`, `R-002`, ... in source order. Columns: `ID | Source | Requirement | Covered by option / stage / step | Status`. `Source` follows the three-form grammar defined in the **Scope provenance** rule above — a free-form `file:line` is not one of the three forms and is rejected. When this run's brief is a fan-out packet, the task manifest's `taskBriefPath` points at that packet file, so `brief:` cites the packet's own headings (`## Scope`, `## Evidence`, `## Requirement Provenance`) — not the headings of the upstream user brief the packet came from. For `covered`, `Covered by` must name the specific Option Candidate and Stage/Step that satisfies it, not just "recommended option". **Enforced:** `validators/validate-run.py` `_validate_requirement_coverage_covered_by` fails a `covered` row whose `coveredBy` is bare "recommended option", names no Option/Stage/Step anchor, or cites a Stage number absent from the Stage Map (whether the cited step *actually satisfies* the requirement remains a worker `DISAGREE(f)` judgment). `Status` is one of `covered`, `gap`, `blocked C-NNN`, or `documented-deviation`. A deviation records the concrete alternative in `coveredBy`, non-empty unique `decisionRefs` (`C-NNN` clarification IDs and/or `D-NNNN` decision-draft numbers), and `approvalDisposition: accepted|blocked C-NNN`. `accepted` is valid only when a referenced clarification is user-confirmed (`answered|resolved` with non-empty `userInput`); `blocked C-NNN` is valid only when that same-report clarification is `open` and `Blocks=approval`. **Enforced:** schema `$defs.ImplementationRequirementCoverageRow` plus `validators/validate-run.py` `_validate_requirement_deviations`; the exact `P-Req-*` queue still comes from `scripts/okstra_ctl/plan_items.py`. If any row is `gap`, plain `blocked C-NNN`, or a deviation whose approval disposition is blocked, the Plan Body Verification gate MUST NOT be `passed` / `passed-with-dissent`; add a matching `Blocks=approval` row for the blocker and keep `approved: false`, and record `coverage-gap` in `planBodyVerification.gateBlockedBy` so the blocking value does not silently read as a worker disagreement. **Exception — no double counting:** a row whose `blocked C-NNN` cites a clarification that *this same run's* §5.5.9 round promoted from a `majority-disagree` item is not an independent blocker; that blocker is already counted as the plan item, and re-counting it makes the run block on a clarification it just authored and carries the row into the next run as a fresh blocker. **Enforced:** `validators/validate-run.py` `_validate_gate_blocked_by` (fails a passing gate with a blocking coverage row) and `_independent_coverage_blockers` (the same-run exclusion).
165
167
  - **Review-rule compliance plan:** when a project-local review rule pack is found, each Option Candidate MUST include the design implication of those rules in its File Structure / interfaces / blast-radius notes. For any helper or data transform used by more than one changed service, the plan must either place it in a shared module or explicitly justify why duplication is intentional. For any test step, the plan must state the observable behavior being asserted, not the internal collaborator call being pinned. For any exported/public method added or renamed, the step must carry the intended noun/side-effect semantics so implementation names can be reviewed before code is written.
166
168
  - the YAML frontmatter MUST include the line `approved: false` (report-writer always emits the unflipped value). The user authorises the next `implementation` run by flipping it to `approved: true` (manual edit or `--approve` CLI). Do NOT recreate any `User Approval Request` body block — the validator fails reports that contain one (see `validators/validate-run.py` deprecated patterns).
@@ -157,7 +157,7 @@ For a `host-text` mapping, render each numbered item as its option label followe
157
157
  - Every CLI-wrapper agent remains `inherit` at the Agent layer because its exact `modelExecutionValue` is applied by `okstra-claude-exec.sh`, `okstra-codex-exec.sh`, `okstra-antigravity-exec.sh`, `okstra-grok-exec.sh`, or `okstra-kimi-exec.sh` according to the assignment provider.
158
158
  - Missing or unsupported family-token mapping is a pre-dispatch contract failure. Never inherit the lead model, choose a nearby alias, or switch provider silently.
159
159
  - Every analysis dispatch sets `name: "<workerId>-worker"`; convergence retries append `-reverify-r<N>`, implementation uses the functional `-executor` / `-verifier` suffix, and report writing uses `report-writer`. These values are retained as `agentName` in session JSONL for usage attribution.
160
- - Every Codex / Antigravity prompt includes `**Pane role:** <functional-role>` so the wrapper's optional fifth argument names both its caller pane and trace pane.
160
+ - Every CLI-worker prompt includes `**Pane role:** <functional-role>` so the entrypoint's optional fifth argument carries the dispatched role. That argument selects the dispatch's idle budget — `executor` and `verifier` run silent build+test suites and get 1500s, every other role 600s — and is recorded in the run's status sidecar. Omitting it defaults to `worker`, i.e. the short budget, which reaps a healthy build mid-suite.
161
161
  - The Agent SDK may supply transport metadata through the in-process worker definition, but the persisted semantic prompt body and primary analysis-packet input remain identical to the CLI-wrapper workers after permitted identity/path normalization.
162
162
  - A retry keeps the same Agent `name`. When logging a twice-failed CLI-wrapper attempt, reference both attempts' `bash_ids` and prompt-history paths.
163
163
  - An internally detected contract violation without a specific worker uses `--agent "claude-lead"` in the error-log event.
@@ -91,7 +91,7 @@ Render every numbered item as its option label followed by its description verba
91
91
  - Worker completion is valid only from `workerDispatches[]`, terminal status sidecars, and required Result Paths. Pane creation alone is not completion.
92
92
  - Reverify uses a fresh jobs file at `runs/<task-type>/state/reverify-jobs-r<N>-<task-type>-<seq>.json`, sets `dispatchKind: "reverify-r<N>"`, and dispatches with `okstra team dispatch --project-root <root> --run-manifest <path> --dispatch-kind reverify-r<N> --jobs-file <jobs-file>`.
93
93
  - Report-writer uses a fresh one-job jobs file with `dispatchKind: "report-writer"` and the same schema, then dispatches through `okstra team dispatch --project-root <root> --run-manifest <path> --jobs-file <jobs-file>`.
94
- - Every reverify or report-writer jobs file carries `workerId`, `provider`, `role`, `modelExecutionValue`, `promptPath`, `resultPath`, `workerResultPath`, and `completionPaths`. For reverify, set `role` to `worker-reverify-r<N>` for the pane title. The report-writer completion paths include both data.json and the worker-results audit file.
94
+ - Every reverify or report-writer jobs file carries `workerId`, `provider`, `role`, `modelExecutionValue`, `promptPath`, `resultPath`, `workerResultPath`, and `completionPaths`. For reverify, set `role` to `worker-reverify-r<N>` — the role selects the dispatch's idle budget and is recorded in the run's status sidecar, so it must name the actual assignment. The report-writer completion paths include both data.json and the worker-results audit file.
95
95
  - After either dispatch, run `okstra team await --project-root <root> --run-manifest <path>` before evaluating terminal status or completion paths.
96
96
 
97
97
  ## Completion, cleanup, and resume
@@ -1,5 +1,20 @@
1
1
  """Bundled Antigravity provider catalog."""
2
+ from typing import Any, Mapping
3
+
2
4
  from okstra_ctl.domain.provider import LeadLaunchSpec, ModelSpec, ProviderSpec
5
+ from okstra_ctl.domain.worker_exec import (
6
+ STREAM_JSON,
7
+ ExecCommand,
8
+ PolicySupport,
9
+ WorkerExecRequest,
10
+ )
11
+ from okstra_ctl.domain.worker_stream import (
12
+ Result,
13
+ StreamEvent,
14
+ Text,
15
+ ToolCall,
16
+ ToolResult,
17
+ )
3
18
 
4
19
 
5
20
  # `agy` needs a tier suffix at dispatch time; model_discovery owns that host
@@ -18,6 +33,138 @@ ANTIGRAVITY = {
18
33
  }
19
34
 
20
35
 
36
+ # The wall-clock cap on one print run, carried over from the wrapper. It is
37
+ # deliberately not derived from the idle budget: an analyser legitimately works
38
+ # for far longer than it goes quiet, so folding the idle budget in here would
39
+ # kill a healthy worker mid-run. Idle is the runner's watchdog to enforce.
40
+ _PRINT_TIMEOUT = "7200s"
41
+
42
+ _STEP_UPDATE = "step_update"
43
+ _RESULT = "result"
44
+ _CALL_STARTED = "ACTIVE"
45
+ _CALL_FINISHED = "DONE"
46
+
47
+
48
+ def step_update_events(event: Mapping[str, Any]) -> tuple[StreamEvent, ...]:
49
+ """Normalise this CLI's stream-json, which shares no key with the other four.
50
+
51
+ Measured 2026-08-11 against agy 1.1.11. Events are keyed on `event` rather
52
+ than `type`, and progress arrives as one `step_update` per step transition:
53
+
54
+ - `step_type: agent_response` carries the worker's prose in `text_delta`
55
+ (absent on the turns that only planned a tool call).
56
+ - `step_type: tool` arrives twice for one call — `ACTIVE` with the
57
+ arguments, then `DONE` with the tool's output.
58
+ - the closing `result` carries `response`, and an `error` string when
59
+ `status` is anything but SUCCESS. A failed run's `response` is empty, so
60
+ without the error text such a run would leave nothing behind at all.
61
+ """
62
+ kind = event.get("event")
63
+ if kind == _RESULT:
64
+ return _result_events(event.get(_RESULT))
65
+ if kind != _STEP_UPDATE:
66
+ return ()
67
+ step = event.get(_STEP_UPDATE)
68
+ if not isinstance(step, Mapping):
69
+ return ()
70
+ if step.get("step_type") == "agent_response":
71
+ text = step.get("text_delta")
72
+ return (Text(body=text),) if isinstance(text, str) and text.strip() else ()
73
+ if step.get("step_type") == "tool":
74
+ return _tool_events(step)
75
+ return ()
76
+
77
+
78
+ def _tool_events(step: Mapping[str, Any]) -> tuple[StreamEvent, ...]:
79
+ info = step.get("tool_info")
80
+ info = info if isinstance(info, Mapping) else {}
81
+ name = str(step.get("tool_name") or info.get("name") or "tool")
82
+ if step.get("state") == _CALL_STARTED:
83
+ return (ToolCall(name=name, detail=_argument(info.get("parameters"))),)
84
+ if step.get("state") != _CALL_FINISHED:
85
+ return ()
86
+ output = str(info.get("output", ""))
87
+ # No outcome field: a `run_command` whose command exited non-zero was
88
+ # measured closing as `DONE` with the failure text in `output` and nothing
89
+ # else to distinguish it, so the outcome is left unreported rather than
90
+ # rendered as success.
91
+ return (ToolResult(body=output, size_bytes=len(output.encode("utf-8"))),)
92
+
93
+
94
+ def _argument(parameters: Any) -> str:
95
+ """The one argument worth a row, without guessing at the parameter names.
96
+
97
+ Parameter keys belong to each tool, not to this CLI — `view_file` names its
98
+ path `AbsolutePath` while `run_command` names its command `CommandLine` —
99
+ so an allowlist of key names would cover only the tools that happened to be
100
+ observed. The first string argument is the subject of every tool measured.
101
+ """
102
+ if not isinstance(parameters, Mapping):
103
+ return ""
104
+ return next(
105
+ (str(value) for value in parameters.values() if isinstance(value, str) and value),
106
+ "",
107
+ )
108
+
109
+
110
+ def _result_events(result: Any) -> tuple[StreamEvent, ...]:
111
+ if not isinstance(result, Mapping):
112
+ return ()
113
+ error = result.get("error")
114
+ failure = (Text(body=error),) if isinstance(error, str) and error.strip() else ()
115
+ response = result.get("response")
116
+ if not isinstance(response, str):
117
+ return failure
118
+ return (*failure, Result(text=response))
119
+
120
+
121
+ class AntigravityExecution:
122
+ """agy CLI invocation.
123
+
124
+ `--add-dir` defines the workspace rather than widening a default one, so the
125
+ project root is included instead of skipped.
126
+
127
+ The prompt is an argument, not stdin: this CLI does not read stdin.
128
+
129
+ `--dangerously-skip-permissions` is required rather than convenient. Headless
130
+ `--print` has nobody to answer a permission request, so without it every
131
+ command tool is auto-denied and the run still reports SUCCESS with an empty
132
+ response.
133
+
134
+ No flag bounds what this worker may write. `--sandbox` was measured on
135
+ 2026-08-10 (agy 1.1.11) and did not block writes outside the `--add-dir`
136
+ workspace through either the shell tool or the file tool — identically to a
137
+ control run without it. It is therefore not passed, and `policy_support`
138
+ reports the gap rather than claiming a boundary that was not observed.
139
+ """
140
+
141
+ def build_command(self, request: WorkerExecRequest) -> ExecCommand:
142
+ argv = ["agy", "--print", request.prompt_text, "--model", request.model]
143
+ for directory in request.policy.write_scope:
144
+ argv += ["--add-dir", str(directory)]
145
+ argv += ["--output-format", "stream-json"]
146
+ argv += ["--print-timeout", _PRINT_TIMEOUT]
147
+ if request.policy.auto_approve:
148
+ argv.append("--dangerously-skip-permissions")
149
+ return ExecCommand(
150
+ argv=tuple(argv),
151
+ stdin_text=None,
152
+ stream_format=STREAM_JSON,
153
+ normalise=step_update_events,
154
+ cwd=request.project_root,
155
+ )
156
+
157
+ def policy_support(self) -> PolicySupport:
158
+ return PolicySupport(
159
+ can_auto_approve=True,
160
+ can_bound_write_scope=False,
161
+ note=(
162
+ "measured 2026-08-10 (agy 1.1.11): --sandbox did not block writes "
163
+ "outside the --add-dir workspace, so no flag bounds this provider"
164
+ ),
165
+ )
166
+
167
+
21
168
  def create_provider() -> ProviderSpec:
22
169
  return ProviderSpec(
23
170
  provider="antigravity",
@@ -32,4 +179,5 @@ def create_provider() -> ProviderSpec:
32
179
  lead_launch=LeadLaunchSpec(
33
180
  executable="agy", model_flag="--model", prompt_flag="--prompt-interactive"
34
181
  ),
182
+ exec_strategy=AntigravityExecution(),
35
183
  )
@@ -1,5 +1,12 @@
1
1
  """Bundled Claude provider catalog."""
2
2
  from okstra_ctl.domain.provider import LeadLaunchSpec, ModelSpec, ProviderSpec
3
+ from okstra_ctl.domain.worker_exec import (
4
+ STREAM_JSON,
5
+ ExecCommand,
6
+ PolicySupport,
7
+ WorkerExecRequest,
8
+ )
9
+ from okstra_ctl.domain.worker_stream import content_block_events
3
10
 
4
11
 
5
12
  CLAUDE = {
@@ -29,6 +36,53 @@ CLAUDE = {
29
36
  }
30
37
 
31
38
 
39
+ # Opening the approval gate for a non-interactive run. Without it this CLI
40
+ # auto-denies any tool call outside the seeded allowlist and the worker silently
41
+ # routes around the denial, which reads as a thinner analysis rather than as a
42
+ # failure. The value is taken from the CLI's own help text; it has not been
43
+ # observed end to end, which is why policy_support claims no write boundary.
44
+ _APPROVAL_ARGS = ("--permission-mode", "bypassPermissions")
45
+
46
+
47
+ class ClaudeExecution:
48
+ """claude CLI invocation.
49
+
50
+ Working directory is the project root even when a stage worktree exists:
51
+ this CLI attaches the worktree with `--add-dir` and runs from the root,
52
+ unlike grok/kimi which run inside the worktree.
53
+ """
54
+
55
+ def build_command(self, request: WorkerExecRequest) -> ExecCommand:
56
+ argv = ["claude", "-p", "--model", request.model]
57
+ for directory in request.policy.write_scope:
58
+ if directory != request.project_root:
59
+ argv += ["--add-dir", str(directory)]
60
+ if request.policy.auto_approve:
61
+ argv += list(_APPROVAL_ARGS)
62
+ # `--verbose` is mandatory, not cosmetic: Claude Code rejects `--print`
63
+ # combined with `--output-format=stream-json` without it and exits 1 in
64
+ # under a second.
65
+ argv += ["--output-format=stream-json", "--verbose"]
66
+ return ExecCommand(
67
+ argv=tuple(argv),
68
+ stdin_text=request.prompt_text,
69
+ stream_format=STREAM_JSON,
70
+ normalise=content_block_events,
71
+ cwd=request.project_root,
72
+ )
73
+
74
+ def policy_support(self) -> PolicySupport:
75
+ return PolicySupport(
76
+ can_auto_approve=True,
77
+ can_bound_write_scope=False,
78
+ note=(
79
+ "the approval flag is taken from help text and has not been "
80
+ "observed end to end; no flag is known to bound this provider's "
81
+ "writes, so none is claimed"
82
+ ),
83
+ )
84
+
85
+
32
86
  def create_provider() -> ProviderSpec:
33
87
  return ProviderSpec(
34
88
  provider="claude",
@@ -52,4 +106,5 @@ def create_provider() -> ProviderSpec:
52
106
  start_session_id_flag="--session-id",
53
107
  resume_session_id_flag="--resume",
54
108
  ),
109
+ exec_strategy=ClaudeExecution(),
55
110
  )
@@ -1,5 +1,11 @@
1
1
  """Bundled Codex provider catalog."""
2
2
  from okstra_ctl.domain.provider import LeadLaunchSpec, ModelSpec, ProviderSpec
3
+ from okstra_ctl.domain.worker_exec import (
4
+ TEXT,
5
+ ExecCommand,
6
+ PolicySupport,
7
+ WorkerExecRequest,
8
+ )
3
9
 
4
10
 
5
11
  CODEX = {
@@ -17,6 +23,40 @@ CODEX = {
17
23
  }
18
24
 
19
25
 
26
+ class CodexExecution:
27
+ """codex CLI invocation.
28
+
29
+ The sandbox stays `workspace-write` even when the approval gate is opened:
30
+ removing the gate is not the same as removing the boundary. A non-interactive
31
+ codex run under the default `on-request` policy blocks on the first approval
32
+ it wants, and with no TTY to answer it ends the turn with exit 0 and no
33
+ output — a success that produced nothing.
34
+
35
+ Working directory is the project root even when a stage worktree exists:
36
+ this CLI attaches the worktree with `--add-dir` and runs from the root,
37
+ unlike grok/kimi which run inside the worktree.
38
+ """
39
+
40
+ def build_command(self, request: WorkerExecRequest) -> ExecCommand:
41
+ argv = ["codex", "exec", "-C", str(request.project_root)]
42
+ for directory in request.policy.write_scope:
43
+ if directory != request.project_root:
44
+ argv += ["--add-dir", str(directory)]
45
+ argv += ["--model", request.model, "--sandbox", "workspace-write"]
46
+ if request.policy.auto_approve:
47
+ argv += ["-c", "approval_policy=never"]
48
+ argv.append("-")
49
+ return ExecCommand(
50
+ argv=tuple(argv),
51
+ stdin_text=request.prompt_text,
52
+ stream_format=TEXT,
53
+ cwd=request.project_root,
54
+ )
55
+
56
+ def policy_support(self) -> PolicySupport:
57
+ return PolicySupport(can_auto_approve=True, can_bound_write_scope=True)
58
+
59
+
20
60
  def create_provider() -> ProviderSpec:
21
61
  return ProviderSpec(
22
62
  provider="codex",
@@ -40,4 +80,5 @@ def create_provider() -> ProviderSpec:
40
80
  "which okstra needs so the lead can reach cmux and start worker CLIs."
41
81
  ),
42
82
  ),
83
+ exec_strategy=CodexExecution(),
43
84
  )