okstra 0.178.0 → 0.179.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/commands/execute/plan-verify.mjs +1 -1
- package/dist/commands/execute/worktree-status.mjs +8 -2
- package/dist/commands/execute/worktree-status.mjs.map +1 -1
- package/dist/commands/lifecycle/install.mjs +1 -1
- package/dist/commands/lifecycle/install.mjs.map +1 -1
- package/dist/commands/report/render-final-report.mjs +3 -3
- package/docs/architecture/storage-model.md +3 -3
- package/docs/architecture.md +10 -9
- package/docs/cli.md +11 -13
- package/docs/for-ai/skills/okstra-inspect.md +3 -3
- package/docs/for-ai/skills/okstra-schedule-gen.md +2 -2
- package/docs/for-ai/skills/okstra-user-response.md +2 -2
- package/docs/project-structure-overview.md +10 -11
- package/docs/task-process/implementation-planning.md +1 -1
- package/docs/task-process/implementation.md +1 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +11 -12
- package/runtime/bin/lib/okstra/globals.sh +2 -2
- package/runtime/bin/lib/okstra/interactive.sh +1 -1
- package/runtime/bin/lib/okstra/usage.sh +11 -9
- package/runtime/bin/lib/okstra-ctl/cmd-rerun.sh +1 -1
- package/runtime/bin/okstra-central.sh +2 -2
- package/runtime/bin/okstra-render-final-report.py +1 -1
- package/runtime/bin/okstra-token-usage.py +1 -1
- package/runtime/prompts/launch.template.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +6 -1
- package/runtime/prompts/lead/context-loader.md +3 -2
- package/runtime/prompts/lead/convergence.md +1 -1
- package/runtime/prompts/lead/okstra-lead-contract.md +4 -4
- package/runtime/prompts/lead/plan-body-verification.md +3 -3
- package/runtime/prompts/lead/report-writer.md +21 -20
- package/runtime/prompts/lead/team-contract.md +1 -1
- package/runtime/prompts/profiles/_common-contract.md +5 -4
- package/runtime/prompts/profiles/_implementation-deliverable.md +1 -0
- package/runtime/prompts/profiles/_implementation-executor.md +2 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +9 -5
- package/runtime/prompts/profiles/implementation.md +4 -4
- package/runtime/prompts/profiles/improvement-discovery.md +2 -2
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +5 -4
- package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +5 -4
- package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +2 -2
- package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +71 -9
- package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +5 -4
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +55 -3
- package/runtime/python/okstra_ctl/analysis_inputs.py +5 -3
- package/runtime/python/okstra_ctl/analysis_packet.py +21 -0
- package/runtime/python/okstra_ctl/backfill.py +12 -5
- package/runtime/python/okstra_ctl/consumers.py +70 -3
- package/runtime/python/okstra_ctl/convergence_engine.py +43 -17
- package/runtime/python/okstra_ctl/dispatch_core.py +47 -11
- package/runtime/python/okstra_ctl/dispatch_state.py +20 -19
- package/runtime/python/okstra_ctl/domain/worker_exec.py +13 -34
- package/runtime/python/okstra_ctl/domain/worker_presentation.py +128 -0
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +5 -0
- package/runtime/python/okstra_ctl/final_report_paths.py +77 -1
- package/runtime/python/okstra_ctl/handoff.py +1 -2
- package/runtime/python/okstra_ctl/implementation_outcome.py +1 -1
- package/runtime/python/okstra_ctl/index.py +4 -4
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +26 -12
- package/runtime/python/okstra_ctl/listing.py +4 -2
- package/runtime/python/okstra_ctl/manager_launch.py +1 -1
- package/runtime/python/okstra_ctl/manager_sync.py +1 -1
- package/runtime/python/okstra_ctl/path_hints.py +2 -2
- package/runtime/python/okstra_ctl/paths.py +13 -10
- package/runtime/python/okstra_ctl/plan_run_root.py +9 -5
- package/runtime/python/okstra_ctl/recap.py +3 -2
- package/runtime/python/okstra_ctl/reconcile.py +3 -1
- package/runtime/python/okstra_ctl/render.py +22 -22
- package/runtime/python/okstra_ctl/report_finalize.py +4 -4
- package/runtime/python/okstra_ctl/rollup.py +1 -1
- package/runtime/python/okstra_ctl/run.py +139 -284
- package/runtime/python/okstra_ctl/run_audit.py +5 -5
- package/runtime/python/okstra_ctl/run_index_row.py +2 -2
- package/runtime/python/okstra_ctl/session_transcript.py +89 -0
- package/runtime/python/okstra_ctl/stage_ledger.py +72 -0
- package/runtime/python/okstra_ctl/stage_map.py +28 -29
- package/runtime/python/okstra_ctl/stage_targets.py +61 -0
- package/runtime/python/okstra_ctl/user_response.py +97 -12
- package/runtime/python/okstra_ctl/wizard.py +43 -65
- package/runtime/python/okstra_ctl/worker_prompt_body.py +6 -7
- package/runtime/python/okstra_ctl/worker_runner.py +76 -213
- package/runtime/python/okstra_ctl/workflow.py +1 -1
- package/runtime/python/okstra_ctl/wrapper_status.py +23 -0
- package/runtime/python/okstra_ctl/write_policy.py +45 -6
- package/runtime/python/okstra_project/state.py +2 -2
- package/runtime/python/okstra_token_usage/__init__.py +1 -1
- package/runtime/python/okstra_token_usage/cli.py +3 -3
- package/runtime/python/okstra_token_usage/report.py +7 -24
- package/runtime/schemas/convergence-groups-v1.0.schema.json +1 -1
- package/runtime/schemas/convergence-groups-v2.0.schema.json +1 -1
- package/runtime/schemas/final-report-v2.0.schema.json +11 -1
- package/runtime/skills/okstra-inspect/facets/history.md +3 -3
- package/runtime/skills/okstra-inspect/facets/recap.md +1 -1
- package/runtime/skills/okstra-inspect/facets/report.md +5 -5
- package/runtime/skills/okstra-inspect/facets/status.md +2 -2
- package/runtime/skills/okstra-pr-gen/SKILL.md +1 -1
- package/runtime/skills/okstra-run/SKILL.md +1 -1
- package/runtime/skills/okstra-schedule-gen/SKILL.md +1 -1
- package/runtime/skills/okstra-user-response/SKILL.md +3 -3
- package/runtime/templates/project-docs/task-index.template.md +1 -1
- package/runtime/templates/report-writer-prompt-preamble.md +1 -1
- package/runtime/validators/forbidden_actions.py +76 -5
- package/runtime/validators/lib/fixtures.sh +14 -10
- package/runtime/validators/lib/runners.sh +1 -1
- package/runtime/validators/validate-implementation-plan-stages.py +3 -0
- package/runtime/validators/validate-report-views.py +1 -1
- package/runtime/validators/validate-run.py +95 -37
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
## File-author ownership (BLOCKING)
|
|
4
4
|
|
|
5
|
-
The final-report data.json is authored by `Report writer worker` when that role is in the roster. The lead reviews
|
|
5
|
+
The final-report data.json is authored by `Report writer worker` when that role is in the roster. The lead reviews the human HTML and the report record but does not write them. Lead-authored fallback is legal only after a real `dispatch_worker` attempt records `error`, `timeout`, or `not-run` with a concrete reason. `release-handoff` remains the intentional single-lead exception.
|
|
6
6
|
|
|
7
|
-
The JSON SSOT path is `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`.
|
|
7
|
+
The JSON SSOT path is `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`. Phase 7 produces the task-specific human HTML sibling from that record. The full reading copy is rendered on demand with `okstra render-final-report <data.json>`. The worker-result pointer at `**Worker Result Path:**` records the record path plus reconciled convergence input. Completion artifacts are the record, the pointer, and the audit sidecar; HTML follows during finalization.
|
|
8
8
|
|
|
9
9
|
New bundles use `schemas/final-report-v2.0.schema.json`. The Markdown keeps verdict, routing, evidence, one structured task deliverable, and audit data for the next agent. The HTML uses `humanSummary`, task `userNarrative`, and structured facts for the user. Raw worker discussion, convergence mechanics, and usage belong to audit structures and never to the HTML human main body.
|
|
10
10
|
|
|
@@ -18,9 +18,9 @@ Emit `frontmatter.approved` as `false` and copy `implementationPlanning.selected
|
|
|
18
18
|
|
|
19
19
|
Emit `frontmatter.approved` as `false` and `frontmatter.implementationOption` as the empty string `""`. The user later flips `approved` to `true` and fills `implementationOption` with the chosen Option Candidate name to authorise and scope the next `implementation` run. Every other report type follows the same empty `implementationOption` default; the schema's non-selected-direction branch requires that field and rejects a selected-direction reference.
|
|
20
20
|
|
|
21
|
-
**As the report-writer worker:** YOU write the
|
|
21
|
+
**As the report-writer worker:** YOU write the report record; the file on disk is the canonical record, so do not return it inline. Do not invoke `okstra render-final-report`.
|
|
22
22
|
|
|
23
|
-
**As the lead:** prepare the report-writer prompt, dispatch the Report writer worker per the Phase 6 dispatch template in okstra-lead-contract.md, and review the
|
|
23
|
+
**As the lead:** prepare the report-writer prompt, dispatch the Report writer worker per the Phase 6 dispatch template in okstra-lead-contract.md, and review the record, the pointer, and the separate audit sidecar in Phase 7. Do not call `write_artifact` against the report paths or worker-result pointer yourself when Report writer worker is in the roster.
|
|
24
24
|
|
|
25
25
|
## When to Use
|
|
26
26
|
|
|
@@ -35,7 +35,7 @@ Emit `frontmatter.approved` as `false` and `frontmatter.implementationOption` as
|
|
|
35
35
|
3. Run `okstra agent-prompt materialize --audience report-writer --assignment-ref initial/report-writer --worker-id report-writer --dispatch-kind report-writer ...`, then run `okstra agent-prompt verify` against the returned `metadataPath`. Use the returned `promptPath` without appending role prose. A correction redispatch repeats this step with a fresh invocation ID and the same audience and assignment reference.
|
|
36
36
|
4. Emit the Phase 6 checkpoint.
|
|
37
37
|
5. For `runner=native-session`, first run `okstra agent-prompt record-dispatch` with the project root, run manifest, metadata path, and `--enforcement-mode host-native-spec-link-gate`, then call the host primitive with only the returned `hostModelValue`. After its result exists, run `okstra agent-prompt link-result` with `--dispatch-id <invocationId>:attempt-1` and the result path before accepting it. For `runner=cli-wrapper`, call `okstra team dispatch` when `terminalBackend` is `cmux-pane`, or `okstra worker-dispatch --workers report-writer` otherwise, which consumes `modelExecutionValue` and verifies the metadata before starting the provider process. Never combine this Phase 6 call with analysis workers.
|
|
38
|
-
6. Call `await_workers([handle])` and verify the data.json Result Path
|
|
38
|
+
6. Call `await_workers([handle])` and verify the data.json Result Path and worker-result pointer at Worker Result Path. Verify the separate heartbeat audit sidecar before accepting the run. **Enforced:** both dispatch adapters keep the two completion paths in `WorkerJob.completion_paths`, and `validators/validate_session_conformance.py` validates the audit sidecar.
|
|
39
39
|
|
|
40
40
|
The complete assignment supplies both runner-specific model values and the prompt header in item 9 below. A native host uses `hostModelValue`; a deterministic provider process uses `modelExecutionValue`; the recorded `**Model:**` header remains the canonical assignment label. Missing or unsupported model resolution is a pre-dispatch contract failure; the common contract does not choose a runtime fallback.
|
|
41
41
|
|
|
@@ -43,7 +43,7 @@ The prompt MUST include, in this order at the top:
|
|
|
43
43
|
|
|
44
44
|
1. `**Project Root:** <absolute-path>`
|
|
45
45
|
2. `**Prompt History Path:** <project-relative-path>` (under current run `prompts/`)
|
|
46
|
-
3. `**Result Path:** runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json` — canonical JSON SSOT. The
|
|
46
|
+
3. `**Result Path:** runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json` — canonical JSON SSOT. The full reading copy is rendered on demand.
|
|
47
47
|
4. `**Worker Result Path:** runs/<task-type>/worker-results/report-writer-worker-<task-type>-<seq>.md` — canonical three-path worker-result pointer and source for the audit-path derivation.
|
|
48
48
|
5. `**Audit sidecar path:** <absolute-path>` — the generated report-writer heartbeat/read-confirmation destination derived from the Markdown `**Worker Result Path:**`, never from Result Path.
|
|
49
49
|
6. `Assigned worker prompt history path: <absolute-path>`
|
|
@@ -56,19 +56,19 @@ The prompt MUST include, in this order at the top:
|
|
|
56
56
|
9. `**Model:** Report writer worker, <modelExecutionValue>` (resolved per Phase 5.5 anchor-header rules)
|
|
57
57
|
10. The full `[Required reading]` clause (see [team-contract](./team-contract.md)) — for Phase 6 it adds two **per-task-type, instruction-set-local** read-only files, both scoped to this run's task-type by `okstra-ctl` at prep time:
|
|
58
58
|
- `<instruction-set>/final-report-schema.json` — a task-type excerpt of schema v2. This is the binding authoring shape; the installed full schema is what the run is judged against. Do **NOT** pull the full repository schema because it is outside the task bundle.
|
|
59
|
-
- `<instruction-set>/final-report-template.md` — the
|
|
59
|
+
- `<instruction-set>/final-report-template.md` — the full reading copy template. It shows the agent-facing ledger shape, not the human presentation. The task-specific HTML renderer reads data.json separately.
|
|
60
60
|
11. The analysis packet path plus a one-line MCP pointer instead of copying the server block verbatim: `**MCP servers:** follow the analysis packet's "Available MCP Servers" section. If the section is absent or says none, treat MCP as unavailable for this run; never infer tools from host configuration.`
|
|
61
61
|
12. `Convergence state: runs/<task-type>/state/convergence-<task-type>-<seq>.json`, followed by pointers to all analysis-worker result files under `worker-results/`. The convergence path is deterministic and is listed even before Phase 5.5 creates the file. Read its classifications (Full/Partial/Contested/Worker-Unique), `roundHistory[]`, `round2SkippedReason`, and `finalClassificationCounts`; populate `crossVerification.roundHistory` in data.json so Section 6 can show which rounds executed, queue sizes, and why Round 2 was (or was not) skipped. The renderer prints the full per-round table only when more than one round ran; single-round or zero-round histories are auto-collapsed to a one-line summary.
|
|
62
62
|
13. `**Report Language:** <en|ko>` — must be either `en` or `ko`; `auto`
|
|
63
63
|
has been resolved by the lead from project.json / global config
|
|
64
64
|
before the dispatch is constructed. The worker copies this verbatim
|
|
65
65
|
into `data.json.meta.reportLanguage`.
|
|
66
|
-
14. An explicit instruction: `You are the author of
|
|
66
|
+
14. An explicit instruction: `You are the author of TWO files: (a) the report record at <Result Path>, and (b) the worker-result pointer at <Worker Result Path>. Maintain the separate heartbeat audit sidecar at <Audit sidecar path>. Do not return the report inline. Do not invoke okstra render-final-report. The dispatch fails when either completion artifact is missing, and session conformance fails when the audit sidecar is missing or invalid.`
|
|
67
67
|
15. The prose budget (dedup contract): `verdictCard.finalConclusion` is the conclusion SSOT — at most 3 sentences. `rationale.*` fields stay within 2 sentences each; `humanSummary` entries stay concise; task `userNarrative` explains each user-facing section once with evidence references. Do not copy these narratives into the AI Markdown. `summary` stays at 3-5 rows unless the run covers multiple tickets. Generation time scales with output volume, so exceeding the budget is a cost bug, not extra diligence.
|
|
68
68
|
|
|
69
|
-
**Fix-run incremental authoring (applies when the run's profile carries a "Fix-Run Carry" block).** Do not author the data.json from scratch. Start by copying the previous run's data.json (the `Previous report` path in the Fix-Run Carry block) to this run's Result Path, then update ONLY the blocks the fix run changed: `meta`/`header` (run seq, dates), `executionStatus`, `implementation.verifierResults`, `implementation.validationEvidence`, `implementation.commitList` / `diffSummary`, `crossVerification`, `verdictCard`, `finalVerdict`, and any `evidence` rows the fix touched. Deliverable prose for unchanged sections is carried forward verbatim — do not re-generate it.
|
|
69
|
+
**Fix-run incremental authoring (applies when the run's profile carries a "Fix-Run Carry" block).** Do not author the data.json from scratch. Start by copying the previous run's data.json (the `Previous report` path in the Fix-Run Carry block) to this run's Result Path, then update ONLY the blocks the fix run changed: `meta`/`header` (run seq, dates), `executionStatus`, `implementation.verifierResults`, `implementation.validationEvidence`, `implementation.commitList` / `diffSummary`, `crossVerification`, `verdictCard`, `finalVerdict`, and any `evidence` rows the fix touched. Deliverable prose for unchanged sections is carried forward verbatim — do not re-generate it. Do not invoke the reading-copy renderer. The schema validation contract is unchanged, so an incrementally-authored data.json passes the same post-hoc gates. The lead's dispatch prompt MUST include the previous data.json path when the carry block is present.
|
|
70
70
|
|
|
71
|
-
**Completion detection after dispatch (BLOCKING).** A dispatch acknowledgement is NOT completion — detect completion via the SSOT protocol in [team-contract](./team-contract.md) "Worker-completion detection", with a pending set covering the data.json (Result Path)
|
|
71
|
+
**Completion detection after dispatch (BLOCKING).** A dispatch acknowledgement is NOT completion — detect completion via the SSOT protocol in [team-contract](./team-contract.md) "Worker-completion detection", with a pending set covering the data.json (Result Path) and worker-result pointer (Worker Result Path). Check the separate audit sidecar before accepting conformance. Do NOT end the turn with a prose "waiting for the report" statement. **Enforced:** the adapters reject a completed transition while any `completionPaths` entry is absent; `validators/validate_session_conformance.py` owns the audit check.
|
|
72
72
|
|
|
73
73
|
### Resume-safe dispatch
|
|
74
74
|
|
|
@@ -91,7 +91,7 @@ Speculative reasons such as "session resume constraint", "runtime state is unava
|
|
|
91
91
|
|
|
92
92
|
## Phase 6 → Phase 7 execution sequence (BLOCKING order)
|
|
93
93
|
|
|
94
|
-
Phase 6 first produces the
|
|
94
|
+
Phase 6 first produces the report record at `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`. Token Usage cells are `null` at this point, and Section 3 does not yet include auto-spawned follow-ups.
|
|
95
95
|
|
|
96
96
|
For an implementation-planning run, the Report writer worker owns the Phase 6 design assessment snapshot: it writes `designPreparation` and every stage's `designSurfaceCoverage` into data.json from the detector output and consolidated plan. It does not create user inputs, consume a user answer as if it were part of that snapshot, or materialize `design-prep-requests/`; `schemas/final-report-v2.0.schema.json` and `validators/validate-run.py` `_validate_design_prep_contract` enforce the snapshot shape, detector coverage, and references.
|
|
97
97
|
|
|
@@ -126,7 +126,7 @@ Phase 7 post-processing is then **one command**. `okstra report-finalize` owns t
|
|
|
126
126
|
okstra report-finalize \
|
|
127
127
|
--project-root <project_root> \
|
|
128
128
|
--run-manifest <runDirectoryPath>/manifests/run-manifest-<task-type>-<seq>.json \
|
|
129
|
-
--report <runDirectoryPath>/reports/final-report-<task-type>-<seq>.
|
|
129
|
+
--report <runDirectoryPath>/reports/final-report-<task-type>-<seq>.data.json
|
|
130
130
|
```
|
|
131
131
|
|
|
132
132
|
Do NOT run the seven steps below by hand. Hand-running them is the recurring root cause of reports shipping with stale activity, `--` token cells, a missing html sibling, Section 3 missing follow-up entries, or Section 4 rows never spawning — the order is load-bearing and a skipped step surfaces only later, as a validator `contract-violated`. Every step is idempotent, so after fixing a reported failure just re-run the same command.
|
|
@@ -135,7 +135,7 @@ The steps it executes, in this contractual order, and the contract each one carr
|
|
|
135
135
|
|
|
136
136
|
1. **`project-activity` — project canonical activity.** Replaces only `agentActivity[]` from this run's canonical events before translation source extraction. A legacy manifest without activity contract v1 is a byte-preserving no-op. Conformance compares IDs, order, and every core field against the canonical events.
|
|
137
137
|
2. **`check-source` — verify the data.json is English.** The same gate as the pre-translator check above, run again here because everything after it derives from the data.json: rendering a Korean SSOT into English chrome, spawning follow-ups from it, and validating it all succeed on a record the next phase cannot read. A failure here means the report-writer authored in the reader's language; re-dispatch it with the English rule rather than editing the data.json by hand.
|
|
138
|
-
3. **`token-usage` — collect usage.** Aggregates `leadUsage` / `workers[].usage` / `usageSummary` into team-state
|
|
138
|
+
3. **`token-usage` — collect usage.** Aggregates `leadUsage` / `workers[].usage` / `usageSummary` into team-state and populates `tokenUsage` and the execution-status usage fields in data.json. It does not render the full reading copy.
|
|
139
139
|
|
|
140
140
|
The data.json paths populated: `tokenUsage.lead.{totalTokens,billableTokens,costUsd}`, the `worker` / `grand` rows, `tokenUsage.cli.costUsd`, and each `executionStatus[].{totalTokens,billableTokens,costUsd,durationMs,cliTotalTokens,cliCostUsd}` for rows whose role matches a team-state worker. The data.json MUST already exist (Phase 6 output).
|
|
141
141
|
|
|
@@ -171,11 +171,11 @@ After `okstra report-finalize` reports `"ok": true`, **execute the run-scoped cl
|
|
|
171
171
|
|
|
172
172
|
## Schema-v2 report data responsibilities
|
|
173
173
|
|
|
174
|
-
The binding authoring shape is the task bundle's `instruction-set/final-report-schema.json`. Populate both audiences in data.json: agent-facing verdict, routing, evidence, task facts, and audits; user-facing `humanSummary` and task `userNarrative`. The
|
|
174
|
+
The binding authoring shape is the task bundle's `instruction-set/final-report-schema.json`. Populate both audiences in data.json: agent-facing verdict, routing, evidence, task facts, and audits; user-facing `humanSummary` and task `userNarrative`. The full reading copy template deliberately omits the full user narrative, while the task-specific HTML deliberately moves `crossVerification`, `executionStatus`, and `tokenUsage` into collapsed audit details.
|
|
175
175
|
|
|
176
176
|
## Legacy schema-v1 Markdown structure reference
|
|
177
177
|
|
|
178
|
-
The remaining numbered-section guide exists only for rendering or diagnosing historical schema-v1 data. New report-writer runs do not author against it; their instruction-set schema and
|
|
178
|
+
The remaining numbered-section guide exists only for rendering or diagnosing historical schema-v1 data. New report-writer runs do not author against it; their instruction-set schema and full reading copy template are authoritative.
|
|
179
179
|
|
|
180
180
|
### Report Header
|
|
181
181
|
|
|
@@ -251,7 +251,7 @@ Token Summary Generation Rules:
|
|
|
251
251
|
|
|
252
252
|
### Implementation-planning section heading contract (schema v1 only)
|
|
253
253
|
|
|
254
|
-
**This does not apply to any run you will author.** New runs are schema v2 (`report_contract.CURRENT_REPORT_SCHEMA_VERSION`), and `validate_phase_boundary` returns before the substring scan when `schemaVersion == "2.0"` — the v2 deliverable is gated by the schema instead, whose `implementationPlanning` block requires every one of these contents as a named key. The v2
|
|
254
|
+
**This does not apply to any run you will author.** New runs are schema v2 (`report_contract.CURRENT_REPORT_SCHEMA_VERSION`), and `validate_phase_boundary` returns before the substring scan when `schemaVersion == "2.0"` — the v2 deliverable is gated by the schema instead, whose `implementationPlanning` block requires every one of these contents as a named key. The v2 full reading copy template carries nine headings and serialises the plan as JSON beneath them, so it cannot produce these strings and is not expected to.
|
|
255
255
|
|
|
256
256
|
Reading this section as a live instruction is a known and expensive mistake: the writer is sent to author headings the v2 template has no place for, and the run reads as structurally unpassable when nothing is wrong with it. It is retained only for rendering or diagnosing historical schema-v1 reports.
|
|
257
257
|
|
|
@@ -330,7 +330,7 @@ You (the report-writer worker) MUST write the worker-result pointer at `**Worker
|
|
|
330
330
|
runs/<task-type>/worker-results/report-writer-worker-<task-type>-<seq>.md
|
|
331
331
|
```
|
|
332
332
|
|
|
333
|
-
Its body contains exactly the project-relative data.json path
|
|
333
|
+
Its body contains exactly the project-relative data.json path and the convergence-state input path. Analysis-worker result files stay in `## Inputs`; do not copy their list into the pointer. **Enforced:** both dispatch adapters include this pointer in `WorkerJob.completion_paths` and refuse `completed` while it is absent.
|
|
334
334
|
|
|
335
335
|
The pointer's frontmatter and header follow `team-contract` "Result Frontmatter" and the standard worker-result header sections. Use `workerId: "report-writer"` and copy the remaining canonical values from `analysis-material.md`; do not duplicate the final-report body.
|
|
336
336
|
|
|
@@ -401,7 +401,7 @@ Every field MUST anchor its claim with at least one evidence reference — a `pa
|
|
|
401
401
|
- **Keep each sentence to one main idea.** A single sentence that stacks four or five clauses with em-dashes and nested parentheticals (300+ characters) is hard to read, and the renderer can only line-break at sentence ends — so break such reasoning into separate sentences. Facts, evidence, and IDs still live in the tables; prose carries only the connective *why*.
|
|
402
402
|
- **Write the report body in English, whatever the Report Language is.**
|
|
403
403
|
The data.json is the SSOT every later phase, validator and agent reads,
|
|
404
|
-
and its
|
|
404
|
+
and its full reading copy has the same audience, so both stay
|
|
405
405
|
in one language. Only the human HTML follows the reader: when
|
|
406
406
|
**Report Language** is not `en`, Phase 7 dispatches the translator
|
|
407
407
|
worker, which writes a sidecar the HTML renderer overlays. You never
|
|
@@ -425,7 +425,7 @@ Every field MUST anchor its claim with at least one evidence reference — a `pa
|
|
|
425
425
|
|
|
426
426
|
Persistence steps that must be performed in Phase 7:
|
|
427
427
|
|
|
428
|
-
- [ ] 1. **Draft
|
|
428
|
+
- [ ] 1. **Draft report record**: Save to `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`
|
|
429
429
|
- [ ] 2. **Update team state**: Update `runs/<task-type>/state/team-state-<task-type>-<seq>.json`
|
|
430
430
|
- Final status, start/end times, and result file paths for each worker
|
|
431
431
|
- Overall run status
|
|
@@ -451,7 +451,8 @@ Persistence steps that must be performed in Phase 7:
|
|
|
451
451
|
|
|
452
452
|
Provide a concise report in the Report Language covering the following:
|
|
453
453
|
- Completion status
|
|
454
|
-
-
|
|
454
|
+
- Human report path (`.html`)
|
|
455
|
+
- Report record path (`.data.json`) and one line to render the full reading copy: `okstra render-final-report <task-qualified data.json>`
|
|
455
456
|
- Team-state path
|
|
456
457
|
- Validator results
|
|
457
458
|
- Resume command path
|
|
@@ -54,7 +54,7 @@ Only workers selected from `recommendedWorkers` in `task-manifest.json` and `res
|
|
|
54
54
|
|
|
55
55
|
0. **Adapter-owned dispatch (BLOCKING).** Every worker start, await, retry, and shutdown goes through the selected runtime adapter. Core state records the outcome but never guesses a host primitive.
|
|
56
56
|
1. The lead is responsible for orchestration, convergence supervision, and final-report review/approval. It never overrides worker analysis and never bypasses a rostered Report writer worker.
|
|
57
|
-
2. `Report writer worker` is NOT an analysis worker. It is excluded from Phase 4/5 (initial analysis) and Phase 5.5 (convergence re-verification). It is spawned only in Phase 6 and is the **author** of the
|
|
57
|
+
2. `Report writer worker` is NOT an analysis worker. It is excluded from Phase 4/5 (initial analysis) and Phase 5.5 (convergence re-verification). It is spawned only in Phase 6 and is the **author** of the report record at `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`.
|
|
58
58
|
3. When `Report writer worker` is in the roster, Lead MUST dispatch it in Phase 6 as a separate invocation after convergence. Omit it from Phase 4/5 analysis selection and pass `--workers report-writer` for a CLI-backed Phase 6 call. The only legal lead-authored fallback is when a dispatch was attempted and recorded a terminal status of `error` / `timeout` / `not-run` with a concrete logged reason. Speculative reasons such as "session resume constraint" or "team is no longer alive" are NOT valid — `dispatch_worker` can start a fresh one-shot assignment through the selected adapter. **Enforced:** `dispatch_core._validate_report_writer_isolation()` rejects every mixed analysis/report plan before process creation, and the default roster selectors exclude `report-writer`.
|
|
59
59
|
4. The assigned model for each role is maintained based on `resultContract.requiredWorkerRoles` in task-manifest.json and the lead model metadata.
|
|
60
60
|
5. Required roles must not be replaced by unnamed generic parallel workers.
|
|
@@ -14,7 +14,7 @@ profile document.
|
|
|
14
14
|
- For a new `implementation-planning` run, the plan-body sequence is initial verification → one planner self-fix → targeted re-verification → user gate. The initial verification is round 1, the targeted re-verification is round 2, and a second automatic self-fix is a contract violation. A user-directed correction does not consume the automatic self-fix limit, and a verification failure after that correction does not restart the automatic loop.
|
|
15
15
|
- **provider-unavailable fallback (tolerance).** A worker dispatch can fail to produce a result for two distinct reasons, and both take the same recovery path. (1) **Pane budget:** the dispatch is rejected because a teammate pane could not be created — this is the harness running out of room for its own teammate panes, not okstra placing a worker. The wording is the host's, so match the condition rather than a fixed string. (2) **Sandbox CLI-start failure:** an external CLI worker wrapper exits non-zero within seconds with empty stdout and its live-log shows `operation not permitted`. In either case the lead spends the one shared retry budget through the assignment's recorded runner. If the provider is still unavailable, record that terminal status and continue only under the convergence quorum rules; never replace it silently with a fixed provider or count a substitute as the original provider's vote. Completed external-CLI workers hold no pane of their own. A pane the harness opened for its own teammate carries no id okstra recorded, so no okstra command closes it — the host and the user own that surface. (This is a prompt instruction, not a code-enforced gate.)
|
|
16
16
|
- Dual-audience final-report contract (shared):
|
|
17
|
-
- data.json is the sole authored report artifact.
|
|
17
|
+
- data.json is the sole authored report artifact. The full reading copy Markdown and human HTML are independently derived from it; neither derived artifact is the other's source. The reading copy is rendered on demand.
|
|
18
18
|
- User-facing information belongs in `humanSummary` and the selected task block's `userNarrative`; it must not exist only in Markdown. The HTML human main body explains the result with those fields plus task facts.
|
|
19
19
|
- Agent coordination and audit details belong in `crossVerification`, `executionStatus`, and `tokenUsage`. HTML may expose them only inside collapsed audit details, never as the primary result.
|
|
20
20
|
- Clarification and approval controls are rendered from data.json IDs. A question that could have been resolved before dispatch through the profile or Reporter Confirmations is an intake failure, not a final-report question.
|
|
@@ -64,7 +64,7 @@ profile document.
|
|
|
64
64
|
- **Enforced:** `validators/validate-run.py` `_validate_open_approval_blocker_provenance` requires `origin` and `userConfirmation` on every open approval blocker and rejects a lead-authored one raised with nobody to ask; `validators/validate_session_conformance.py` `_check_user_confirm_checkpoints` requires the matching `PROGRESS: user-confirm <C-NNN>` line for every row the report claims the user was asked about.
|
|
65
65
|
- This contract is the single authority on brief consumption. Phase-specific addenda may *tighten* these rules but may not relax them.
|
|
66
66
|
- Clarification request policy (shared — applies whenever a profile uses `## 1. Clarification Items`):
|
|
67
|
-
- Schema-v2 final reports author `clarificationItems[]` in data.json; task-specific HTML renders the question and response controls directly from those IDs, and
|
|
67
|
+
- Schema-v2 final reports author `clarificationItems[]` in data.json; task-specific HTML renders the question and response controls directly from those IDs, and the full reading copy renders the same array as one headed section per row. The remaining table-layout rules describe schema-v1 compatibility and analysis-worker result tables only.
|
|
68
68
|
- **Every row that is still `open` and carries `Blocks=approval` records two more fields.** Withholding approval is the most expensive thing a report does to a run, and until these fields existed a blocker could not be told apart from a question nobody had put to the user.
|
|
69
69
|
- `origin` — who raised it. `worker-finding` (an analyser or verifier reached it on its own evidence), `material-gap` (neither the brief nor the codebase answers it), or `lead-directed` (the lead's own judgment, **including anything the lead instructed a worker to raise**). A lead that seeds its conclusion into a worker prompt and then reports the worker's agreement as an independent finding has mislabelled the row; that shape is what let one run block on a question its own lead had authored.
|
|
70
70
|
- `userConfirmation` — what happened before the row was written. `asked-and-answered`, `asked-awaiting` (asked, no answer yet), or `deferred-no-interactive-session` (this run had no user to ask). Record an answer in `userInput` and move `status` to `answered`.
|
|
@@ -84,13 +84,14 @@ profile document.
|
|
|
84
84
|
- **One decision per row.** A `decision` row asks one question. When a single option bundles two independent decisions — ones the user could answer differently — split the row, even though each option in it is individually pickable. Bundling forces the user to buy the expensive half in order to get the cheap half, and the answer then records agreement to something they were never asked about. The tell is usually the option's own `scopeImpact`: an option that has to claim `cross-repo` because *one* of the two things it bundles reaches another repository is carrying two decisions of very different cost. Whether two clauses are one decision or two is a judgement, so no validator checks it; this rule and the §5.5.9 adversarial round are the enforcement, and a verifier citing the `scopeImpact` mismatch is what makes a `DISAGREE` on it concrete rather than a matter of taste.
|
|
85
85
|
- each row's `Blocks` column picks one of `{approval, next-phase, none}`. `approval` is reserved for items that gate an approval action, especially the `implementation-planning` `approved:` frontmatter flip; outside `implementation-planning`, unresolved brief reporter-confirmation rows use `next-phase` instead. `next-phase` blocks the next run from starting cleanly. `none` is informational/audit-only.
|
|
86
86
|
- write every entry in full, descriptive sentences that a non-developer can act on without further context. Avoid abbreviations and internal jargon. The `Statement` cell must state *what* is needed, *why* the answer / attachment changes the next step, and (for `material`) *where* the user can find it and *where* to place it. The `Expected form` cell must state the answer shape (yes/no, one of the options, number/date, file path, short description, etc.); supply concrete option choices when applicable.
|
|
87
|
+
- **Record coordinates only.** A clarification `statement`, `expectedForm`, or `options[]` answer/rationale may cite a report-record row id (`RB-002`, `C-014`) or a `path:line`. Do not cite a section number (`§4.7`, `§1`). That number exists only on one full reading copy. **Enforced:** `validators/validate-run.py` `_validate_clarification_record_coordinates`.
|
|
87
88
|
- **Schema-v2 authors do not use the string grammar below.** A v2 `Kind=decision` row carries its choices in `options[]` (see the Clarification recommendation fragment for the field list); the renderer and the `okstra user-response` picker both build their selectable options from that array, so a choice that exists only in prose is a choice the user cannot pick. The rest of this bullet governs schema-v1 tables and analysis-worker result tables, which have only string cells.
|
|
88
89
|
- if a schema-v1 table or an analysis-worker result table requires a recommended answer, alternatives, or an evidence-check note, encode it inside the existing 4-column schema: put evidence notes in `Statement` as `Evidence checked: <path:line>` or `Evidence checked: none — <human-only reason>`, and put recommendations/options in `Expected form` as `Recommended: (a) <answer> — <rationale>; Alternatives: (b) <option> (c) <option>`. The recommended answer is always the first option and MUST carry the `(a)` label; alternatives continue the same letter sequence from `(b)` (a lone alternative is `(b) <option>`, never restart at `(a)`), so the full option set reads `(a) (b) (c) …` in order and renders each as its own selectable option. Do **not** append a pick-one answer-space summary such as `(pick 1 of A / B)` or `(pick N of …)` to `<options>` — the rendered `<select>` already enforces single choice, and that annotation leaks verbatim into an option label. Do not add `Recommended`, `Evidence`, `Alternatives`, or `evidence-checked` columns, and do not break the merged record-meta cell back into separate columns.
|
|
89
90
|
- For schema v2, data.json is canonical and the HTML exports answers to a user-response sidecar; the source report is never edited. `--resume-clarification` carries those answers into the next run. The lower-level `--clarification-response <path>` remains available for scripted runs.
|
|
90
|
-
- When a response is carried in, reconcile every prior `clarificationItems[]` row against new evidence and update its status to `resolved` or `obsolete` before issuing the next verdict. Schema-v1 compatibility Markdown may additionally render its conditional Section 0; schema-v2
|
|
91
|
+
- When a response is carried in, reconcile every prior `clarificationItems[]` row against new evidence and update its status to `resolved` or `obsolete` before issuing the next verdict. Schema-v1 compatibility Markdown may additionally render its conditional Section 0; the schema-v2 full reading copy records decisions under `## Clarification and User Decisions`.
|
|
91
92
|
- **Supersession (BLOCKING).** Reconciling the `C-*` row is only half of incorporating an answer. An answer does not merely *add* a decision — it *invalidates* whatever the previous run wrote under the opposite assumption. Before issuing the next decision, walk the prior deliverable prose for every statement the answer makes false and **delete or rewrite it**, then record the retirement. Adding the new decision while leaving the contradicting sentence in place puts two opposite instructions for the same symbol in one document; the implementer must then guess which is live, and the next verification round correctly blocks on it. In `implementation-planning` this record is `implementationPlanning.supersessionLedger[]` — one entry per answered clarification, either `disposition: superseded` (with the retired statement, its replacement, and the sections revised) or `disposition: no-dependent-statement` (with a rationale). **Enforced:** `validators/validate-run.py` `_validate_supersession_ledger` requires an entry per answered clarification; whether the claim is *true* is what the §5.5.9 adversarial round tests.
|
|
92
93
|
- Verdict Card data consistency (shared; schema-v1 Markdown keeps the legacy visible card):
|
|
93
|
-
- The Card carries no verdict token — the token lives once, in `finalVerdict.verdictToken`, and every gate reads it there. `verdictCard.direction` MUST byte-match `finalVerdict.direction`; next-step routing must agree with `recommendedNextSteps[0]`. The
|
|
94
|
+
- The Card carries no verdict token — the token lives once, in `finalVerdict.verdictToken`, and every gate reads it there. `verdictCard.direction` MUST byte-match `finalVerdict.direction`; next-step routing must agree with `recommendedNextSteps[0]`. The full reading copy and human summary are derived from the data fields without repeating both visible sections. **Enforced:** `validators/validate-run.py` `_validate_verdict_card_fields`.
|
|
94
95
|
- Cross-worker traceability (shared — applies to every analysis worker output and to the lead's `## 6.` / `## 2.` tables in the final-report):
|
|
95
96
|
- **Worker-side item IDs (free-form but unique within the worker).** Every row item in sections 1–5 (and any optional section 6) of an analysis worker's output MUST carry an item ID that is unique within that one worker's result file. The ID convention is the worker's choice — `F-001` / `F-002` per the suggested schema, `1.1` / `1.2` / `1.3` as Codex tends to use, or any other shape — but it MUST appear as the leading column of the row (for table-form items) or as a `[<ID>]` prefix (for bullet/numbered items). Workers that emit findings without IDs make cross-worker reconciliation impossible.
|
|
96
97
|
- **Lead-side ID assignment + source preservation.** When the lead (or `report-writer-worker`) synthesises consensus, difference, or primary-evidence rows from worker outputs, the lead assigns a fresh `C-NNN` / `D-NNN` / `E-NNN` row ID. Each `sourceItems` field MUST list every contributing worker:item pair (e.g. `claude:F-001`, `codex:1.1`, `grok:F-3`, `kimi:2.4`) so an agent can trace the synthesised row to the worker result. Bare worker names are rejected. **Enforced:** `schemas/final-report-v2.0.schema.json` `$defs.SourceItem` pins each entry to `^[a-z][a-z-]*:[A-Za-z0-9._-]+$`, and `ConsensusRow` / `PrimaryEvidenceRow` require non-empty `sourceItems`.
|
|
@@ -62,5 +62,6 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
62
62
|
- **A `FAIL` synthesised verdict withholds the two writes below.** They are what marks the stage `done`, so performing them on a stage whose verifier found a blocking defect stacks the next stage on a confirmed regression. When the synthesised verdict is `FAIL`: write NO carry sidecar, and append a `status:"failed"` row in place of the `done` row — same `okstra_ctl.consumers.append_consumer` call, carrying `report_path` and the SHA of HEAD. That row is terminal *without* completion: dependent stages stay blocked because this stage is not done, while its worktree-registry occupancy is released so a fix run can re-enter the same stage number — `--stage <N>` reuses the preserved worktree and branch instead of provisioning a new one. State the reason in the report's `Stage sidecar evidence` section as `withheld`. **Enforced:** `validators/validate-run.py` `_validate_stage_carry_sidecar_exists` accepts a missing carry file only when that field is non-empty, so silently skipping the sidecar still fails the run.
|
|
63
63
|
- On a non-`FAIL` verdict, for this run's single stage: write its JSON verbatim to `runs/<impl-task-key>/carry/stage-<N>.json`. Refuse to overwrite an existing file (one stage = one sidecar; a fix run re-entering after a `failed` row writes the first one, because a withheld stage never wrote it).
|
|
64
64
|
- On a non-`FAIL` verdict, for this run's single stage: append a `status:"done"` row to `runs/<plan-task-key>/consumers.jsonl` with `completed_at`, `carry_path`, `report_path` (this run's final-report path relative to the run root), and the SHA of HEAD. Append it with `okstra_ctl.consumers.append_consumer` (NOT a raw filesystem write) — that call honours the consumers lock AND releases this stage's worktree-registry occupancy, so later runs stop seeing a finished stage as a concurrent run. `report_path` lets `final-verification` cite each stage's originating report when assembling its Source Implementation Report list.
|
|
65
|
+
- **Reopening a settled stage.** Appending a `status:"failed"` row after a `done` one withdraws that stage: `--stage N` accepts it again, its dependents stop resolving a base from the withdrawn head, and whole-task final-verification blocks until it is settled again. Use the same `okstra_ctl.consumers.append_consumer` call with a `reason`, and re-run the stage rather than repairing the tree outside okstra — a stage reverted outside the ledger leaves the recorded `head_commit` and the carry sidecar naming a tree that no longer exists, and the next stage branches from it.
|
|
65
66
|
- The verifier round, Phase 5.5 convergence, and this Phase 6 report run **once per run** over this stage's diff — NOT per step.
|
|
66
67
|
- Quote this stage's new contents (the sidecar JSON in full and the new consumers row by itself) in the final report's `Stage sidecar evidence` deliverable section.
|
|
@@ -85,7 +85,8 @@ persisted prompt lacks the heading `Coding-conventions preflight`
|
|
|
85
85
|
|
|
86
86
|
## Allowed actions during the run
|
|
87
87
|
|
|
88
|
-
- **Edit / Write on approved project source files**: scope is bounded first by the shared Resource boundary, then by the approved plan's file list. Editing files outside the plan's list is permitted only when strictly needed to satisfy a step, and MUST be recorded in
|
|
88
|
+
- **Edit / Write on approved project source files**: scope is bounded first by the shared Resource boundary, then by the approved plan's file list. Editing files outside the plan's list is permitted only when strictly needed to satisfy a step, and MUST be recorded in your worker result's `Out-of-plan edits` block with rationale.
|
|
89
|
+
- **The block's shape is fixed, because a machine reads it (BLOCKING).** Write a `## Out-of-plan edits` heading, then one `- ` line per file whose first backticked token is the repository-relative path, followed by the rationale: ``- `src/http/router.ts` — the plan's step 3 needs a route the plan did not list``. Nothing else on that line is read. With no such edits, write the heading and the single line `- (none)`. The write audit compares the stage's changed files against the plan's paths union this block: a file edited outside the plan and absent here settles the run as a contract violation, and a path this parser cannot find is the same as one you never declared.
|
|
89
90
|
- read-only inspection commands: `git status`, `git diff`, `git log`, `grep`, `rg`, `find`, `cat`, `ls`, file Read tools
|
|
90
91
|
- build, lint, type-check, and test commands (`npm test`, `pytest`, `go build`, `cargo test`, `bash -n`, etc.)
|
|
91
92
|
- **local git operations only**: `git add`, `git commit`. Prefer small commits keyed to plan steps.
|
|
@@ -219,7 +219,7 @@ A mocked unit test cannot observe the SQL a query builder actually emits — `co
|
|
|
219
219
|
|
|
220
220
|
## All-verifier-failure policy
|
|
221
221
|
|
|
222
|
-
If every verifier present in the resolved roster ends with a non-result terminal status (`timeout`, `error`, `not-run`) — i.e. zero independent verdicts were produced — the run MUST end with status `blocked` and route to a follow-up `error-analysis` run. The Okstra lead MUST NOT substitute its own verdict in place of the missing verifier outputs; synthesis requires at least one independent verifier's verdict. If one or more verifiers fail but at least one returns a verdict, the run proceeds with the surviving verdict(s) and the final report MUST explicitly notate which verifiers were unavailable, with the captured error / timeout evidence per failed verifier.
|
|
222
|
+
If every verifier present in the resolved roster ends with a non-result terminal status (`timeout`, `error`, `not-run`) — i.e. zero independent verdicts were produced — the run MUST end with status `blocked` and route to a follow-up `error-analysis` run. The Okstra lead MUST NOT substitute its own verdict in place of the missing verifier outputs; synthesis requires at least one independent verifier's verdict. If one or more verifiers fail but at least one returns a verdict, the run proceeds with the surviving verdict(s) and the final report MUST explicitly notate which verifiers were unavailable, with the captured error / timeout evidence per failed verifier. Record such a verifier's `verifierResults[].verdict` as `not-run`, never as `FAIL`: `FAIL` is a rejection of the diff and it drives two gates — it blocks a passing verdict and it opens a stage fix cycle — so spending it on a verifier that produced no verdict manufactures a defect that does not exist.
|
|
223
223
|
|
|
224
224
|
## Verifier-specific forbidden actions (any occurrence → terminal status `contract-violated`)
|
|
225
225
|
|
|
@@ -63,6 +63,10 @@ roles:
|
|
|
63
63
|
- **directive-first ambiguity resolution** (the same rule, pointed at the user instead of the code): any ambiguity the run's directive, the brief, the carried-in `user-responses/` sidecars, or the user's in-session instruction already answers MUST be resolved that way and recorded with the quoted instruction. Writing a clarification row for something the user already decided is the same defect as writing one for something the code already answers — and it costs more, because the row withholds approval until a whole separate answer cycle closes it. When an instruction points at a document, treat every item in that document as decided, including the ones the document itself flagged as needing a decision (shared rule: `_common-contract.md` "User instruction outranks the material it points at").
|
|
64
64
|
- flag any requirement that is ambiguous, contradictory, or missing success criteria — register each one as a row in the report's `## 1. Clarification Items` table with `Blocks=approval` instead of guessing
|
|
65
65
|
- read `<PROJECT_ROOT>/.okstra/glossary.md` and `<PROJECT_ROOT>/.okstra/decisions/` titles if present. Absent okstra memory files are the normal state — do not error. Treat the brief's `terminology:*` resolutions from `requirements-discovery` (if any) as authoritative; if missing, resolve any remaining fuzzy term as a `Blocks=approval` clarification row.
|
|
66
|
+
- **Stage Ledger (read before drafting the Stage Map):** when this task already has a plan on disk, the analysis packet carries a `## Stage Ledger` JSON block listing every stage with its `status` (`done` / `active` / `ready` / `blocked`), `dependsOn`, and done commit. It states what exists, not what to plan. Two rules follow from it:
|
|
67
|
+
- A stage whose `status` is `done` is already implemented and will not be executed again. Carry its plan body forward as written; do not rewrite its steps, and do not fold its work into a new stage.
|
|
68
|
+
- Every stage number in the ledger is taken. A new stage takes the next number after the highest one listed; numbers are never reused or reordered. **Not yet machine-enforced** — the validator for this rule lands with the plan-amendment feature.
|
|
69
|
+
- The block is absent on a task's first planning run. Its absence means there is no prior plan, not that no stage is done.
|
|
66
70
|
- Primary focus areas:
|
|
67
71
|
- requirement gaps
|
|
68
72
|
- affected components and boundaries
|
|
@@ -92,7 +96,7 @@ roles:
|
|
|
92
96
|
- **Planner-only test surface:** add `manual-user-test` only when a test prerequisite changes the implementation interface or acceptance contract; the V1 detector never emits it. **Enforced:** `validators/validate-run.py` `_validate_detector_coverage` rejects detector-produced `manual-user-test`, `schemas/final-report-v2.0.schema.json` permits its planner-authored shape, and `prompts/lead/plan-body-verification.md` verifies the stage-action rationale.
|
|
93
97
|
- **Trivial task:** when the detector returns no surfaces and no interface/acceptance-changing manual test input exists, `designPreparation` MUST use `mode: no-design-inputs`, an empty `items` array, and a concrete reason tied to the plan. **Enforced:** `schemas/final-report-v2.0.schema.json` `$defs.DesignPreparation` requires the reason and empty array for that mode; `validators/validate-run.py` `_validate_design_prep_contract` validates the marked V1 payload.
|
|
94
98
|
- Approval gate (phase-specific addendum to shared authority rule):
|
|
95
|
-
- The
|
|
99
|
+
- The report record `frontmatter.approved` field is the only authorised approval gate. report-writer always emits `false`. The user clears it by invoking the next phase with `--approve`, or by confirming approval in the in-session wizard. Editing the full reading copy does not approve the plan. `okstra_ctl.run._validate_approved_plan` reads this field and refuses entry until it is `true`.
|
|
96
100
|
- Cross-verification mode:
|
|
97
101
|
- Phase 5.5 finding convergence runs in **adversarial mode** for this phase (`convergence.adversarial=true`). Verifiers actively try to refute each worker finding (requirement gap / risk / plan item) by re-inspecting its cited evidence; the burden of proof sits on the claim. See `prompts/lead/convergence.md` §"Adversarial Verification Mode".
|
|
98
102
|
- §5.5.9 plan-body verification runs with an **adversarial posture** (`prompts/lead/plan-body-verification.md` §"Adversarial plan-body posture"): verifiers open and confirm every cited path / command and put the burden of proof on the plan. The gate threshold is majority-based for kinds `b`/`c`/`e`, but a single `DISAGREE` blocks on its own for the concrete, safety-critical kind `a` (path/symbol mismatch) — and `f` on `P-Req-*` items. `P-Var-*` items are excepted from the kind-`a` exception: a variation-point defect takes a majority. Rollback ordering (`d`) is advisory and never blocks the gate — a rollback is executed by a human, not by okstra's workers or verifiers. A majority also needs ≥2 participating votes, so a lone dissent whose peer returned a non-result does not block on a majority-gated kind (see that contract's §"Adversarial plan-body posture").
|
|
@@ -118,7 +122,7 @@ roles:
|
|
|
118
122
|
- For a selected-direction plan, the plan-ready schema branch requires `planningContract`, `outcome`, `selectedDirectionRef`, `directionRealization`, `stageMap`, `stages`, `designPreparation`, `dependencyMigrationRisk`, `validationChecklist`, `rollbackStrategy`, `requirementCoverage`, `coverageSummary`, `variationPointAnalysis`, and `planBodyVerification`. Its `direction-invalidated` branch contains no execution fields.
|
|
119
123
|
- Each `stages[]` entry requires `stage`, `title`, `sliceValue`, `acceptance`, `carryIn`, `stepwiseExecution` (1–6 rows), `exitContract`, and `stageValidation`. Each `stageMap[]` row requires `stage`, `title`, `dependsOn`, `stepCount`, `exitContractSummary`.
|
|
120
124
|
- Beyond the schema, `validators/validate-run.py` reads the same data.json for `_validate_planning_conformance_declared`, `_validate_end_state_coverage`, `_validate_requirement_provenance`, `_validate_stage_has_requirement`, and `_validate_plan_body_state_file`. These run for every planning report regardless of schema version.
|
|
121
|
-
- **Do not chase English heading substrings.** `PLANNING_REQUIRED_SECTIONS` and the Markdown scan in `collect_validation_errors` live inside `validate_phase_boundary`, which returns immediately when `schemaVersion == "2.0"` — they gate historical v1 Markdown only. The v2
|
|
125
|
+
- **Do not chase English heading substrings.** `PLANNING_REQUIRED_SECTIONS` and the Markdown scan in `collect_validation_errors` live inside `validate_phase_boundary`, which returns immediately when `schemaVersion == "2.0"` — they gate historical v1 Markdown only. The v2 full reading copy template renders nine headings and serialises the plan as JSON beneath them, so those substrings cannot appear, and a report is not defective for lacking them.
|
|
122
126
|
- Per-stage vertical slice and TDD contract (BLOCKING — enforced on the data, not on heading tokens):
|
|
123
127
|
- Every stage declares `sliceValue`, `acceptance`, and the three cases `testCaseSuccess` / `testCaseBoundary` / `testCaseFailure` — happy path, edge/boundary input, failure input. **Enforced:** the v2 schema's `if not tddExemption then require` conditional on `ImplementationPlanStage`.
|
|
124
128
|
- The first `stepwiseExecution` row's `action` starts with `RED:` and its `expected` reads FAIL; some later row's `action` starts with `GREEN:` and its `expected` reads PASS. **Enforced (S10c):** `collect_data_validation_errors` in `validators/validate-implementation-plan-stages.py`, run from `validate-run.py` `_append_stage_data_failures`.
|
|
@@ -157,9 +161,9 @@ roles:
|
|
|
157
161
|
**Pick the cases by distinct outcome, not by line coverage.** When the stage writes or reconciles state, its meaningfully different outcomes are usually more than three — normal success, target already in the desired state (resume), existing data reused rather than created, a conflicting concurrent state, target absent, and mid-way failure with rollback. Enumerate the ones this stage actually implements and route them across the three lines (the `boundary` line is where resume / already-done / reuse belongs; `failure` carries conflict, absence, and rollback), naming each in the cell rather than collapsing them into "edge input". An implemented outcome with no declared case is a coverage gap the executor will not backfill.
|
|
158
162
|
- **Per-stage subsections** (`## 5.5.<i> Stage <i>: <title>` for each `i`), each containing the four required subsections:
|
|
159
163
|
- `### Carry-In` — for `depends-on (none)`: task-brief only. Otherwise: each depended-on stage's static exit contract + runtime sidecar path `runs/<impl-key>/carry/stage-<i>.json` placeholder.
|
|
160
|
-
- `### Stepwise Execution Order` — bite-sized table with `step | action | files | command | outcome | expected`. `outcome` is one word — `PASS` or `FAIL` — and `expected` is the sentence saying what that looks like here; a verdict written inside the sentence is not read as one. The `files` cell lists each touched path in full and `<PROJECT_ROOT>`-relative — never ellipsis-abbreviated (`…` / `...`), which does not resolve and is rejected by plan-body verification as a kind-b path mismatch. **Effective row count ≤ 8** (excluding header / divider / blank). Each step is one cohesive, self-contained change (no lower time bound; it may span several files that change together); for code steps include actual code or diff sketch. **TDD ordering is MUST, not a preference:** the **first** effective step's `action` cell MUST start with the literal `RED:` and describe the failing test(s) that capture this stage's `Acceptance` **and the three declared `Test case (success|boundary|failure)` lines** (`outcome` = `FAIL`) — the RED step encodes the case set, not a single happy-path assertion; at least one later `action` cell MUST start with the literal `GREEN:` and describe the minimal implementation that makes it pass (`outcome` = `PASS`); an optional refactor step starts with `REFACTOR:`. **Exemption:** doc-only / config-only / pure-rename stages with no observable runtime behaviour may omit RED/GREEN by declaring one line `TDD exemption: <reason>` in the stage section (mirrors the executor's per-step exemption in `_implementation-executor.md`). Validator S10c enforces RED-first + GREEN; the `outcome` cell agreeing with its `RED:` / `GREEN:` prefix is a schema conditional (`StageStepRow.allOf`), so a plan whose data.json says otherwise never reaches the validator. S10e rejects a `TDD exemption:` whose reason is not one of doc-only / config-only / pure-rename (both in `validators/validate-implementation-plan-stages.py`).
|
|
164
|
+
- `### Stepwise Execution Order` — bite-sized table with `step | action | files | command | outcome | expected`. `outcome` is one word — `PASS` or `FAIL` — and `expected` is the sentence saying what that looks like here; a verdict written inside the sentence is not read as one. The `files` cell lists each touched path in full and `<PROJECT_ROOT>`-relative — never ellipsis-abbreviated (`…` / `...`), which does not resolve and is rejected by plan-body verification as a kind-b path mismatch. **The data.json row additionally carries `plannedPaths`: the same paths as an array, one repository-relative path per entry, with no globs, exclusions, counts or commentary.** `files` is the sentence a reader sees; `plannedPaths` is the ledger the implementer write policy enforces against, and a step whose paths live only in the prose cell fails the run with every file it touched read as an unauthorized source change. When a step legitimately covers a set too large to enumerate, split it or name the directory the set lives under — the policy compares against these entries and nothing else. **Effective row count ≤ 8** (excluding header / divider / blank). Each step is one cohesive, self-contained change (no lower time bound; it may span several files that change together); for code steps include actual code or diff sketch. **TDD ordering is MUST, not a preference:** the **first** effective step's `action` cell MUST start with the literal `RED:` and describe the failing test(s) that capture this stage's `Acceptance` **and the three declared `Test case (success|boundary|failure)` lines** (`outcome` = `FAIL`) — the RED step encodes the case set, not a single happy-path assertion; at least one later `action` cell MUST start with the literal `GREEN:` and describe the minimal implementation that makes it pass (`outcome` = `PASS`); an optional refactor step starts with `REFACTOR:`. **Exemption:** doc-only / config-only / pure-rename stages with no observable runtime behaviour may omit RED/GREEN by declaring one line `TDD exemption: <reason>` in the stage section (mirrors the executor's per-step exemption in `_implementation-executor.md`). Validator S10c enforces RED-first + GREEN; the `outcome` cell agreeing with its `RED:` / `GREEN:` prefix is a schema conditional (`StageStepRow.allOf`), so a plan whose data.json says otherwise never reaches the validator. S10e rejects a `TDD exemption:` whose reason is not one of doc-only / config-only / pure-rename (both in `validators/validate-implementation-plan-stages.py`).
|
|
161
165
|
- **The `command` cell runs inside an okstra task worktree, not a bare checkout (BLOCKING).** okstra provisions `.okstra`, the configured sync entries (`.project-docs`, `.claude`, …), and — for `implementation` — a nested `stage-<N>/` worktree into the tree the step executes in. Two consequences bind every command you write:
|
|
162
|
-
- **Clean-tree assertions use `okstra worktree-status --check-clean`.** A bare `git status --porcelain` is never empty there, so an assertion built on one fails on okstra's scaffolding rather than on the stage's work. The okstra command asks the same question over source paths only and exits 1 when dirty, so it
|
|
166
|
+
- **Clean-tree assertions use `okstra worktree-status --check-clean`.** A bare `git status --porcelain` is never empty there, so an assertion built on one fails on okstra's scaffolding rather than on the stage's work. The okstra command asks the same question over source paths only and exits 1 when dirty, so it stands alone as a step's assertion: `okstra worktree-status --check-clean`. Validator S13 rejects the bare form. Do not add a `git tag stage-<N>-exit` to the step — okstra writes that tag itself when it settles the stage, at the commit the carry evidence records, and a step that tags mid-stage puts it on an earlier commit.
|
|
163
167
|
- **Never read an `.okstra/` artifact back out of a git object.** `.okstra/**` is gitignored and never committed — the executor aborts a commit that stages an ignored path and the verifier reports a committed `.okstra` path as a branch defect — so `git cat-file -e <tag>:.okstra/…`, `git show <tag>:.okstra/…`, and every variant of that read can never resolve, at any tag, in any stage. A later stage that needs a QA artifact reads it from the working tree or receives it through the carry sidecar / verifier result; do not design a stage contract around one being reachable from a tag. Validator S12 rejects the read.
|
|
164
168
|
- **Per-stage conformance declaration (mandatory one line, in the stage section — same placement freedom as `TDD exemption:`):** the stage MUST carry exactly one of:
|
|
165
169
|
- `Conformance tests: stage-<N> — <task_root>/qa/scripts/stage-<N>.<ext> (requires=[db|io|http|external,...])` — a Tier3 verification script that proves this stage's upstream requirements (brief / requirements-discovery / error-analysis / improvement-discovery → this stage's `Acceptance`) hold against **real** DB rows, real endpoints, or the real external API — NOT mocks. When you emit this line you MUST also (a) write the script to `<task_root>/qa/scripts/stage-<N>.<ext>` and (b) add a matching entry to `<task_root>/qa/conformance-manifest.json` with fields `stageKey` (= `<task-id>-stage-<N>`), `script`, `runCommand`, `requirementIds`, `requires` (subset of `{db, io, http, external}`), `passContract`, `exemption: null`, `waiver: null`. The script's standard interface: a `main` that exits `0`=PASS / non-zero=FAIL, and whose stdout ends with `QA-RESULT: PASS|FAIL` followed by one `REQ <id>: PASS|FAIL: <reason>` line per requirement. When the verification body is a test spec, author it with the project's own test framework (devDependency) invoked via a discovery override at `<task_root>/qa/scripts/` (jest: `--config <project config> --roots <task_root>/qa/scripts`) — never hand-roll `describe`/`expect` and never widen the project's own test config; for TypeScript specs also write `<task_root>/qa/scripts/tsconfig.json` extending the project tsconfig with the runner's `types` entry so editors resolve the file.
|
|
@@ -206,7 +210,7 @@ roles:
|
|
|
206
210
|
- **Requirement Coverage (mandatory, §5.5.8):** selected-direction plans preserve the original requirement IDs and link each row to `stageRefs`, `stepRefs`, `validationRefs`, and `fileRefs`; exact forward and reverse coverage is enforced by `validate_selected_direction_plan`. Legacy candidate-comparison plans retain one `R-NNN` row per concrete requirement and the existing Option Candidate plus Stage/Step `coveredBy` semantics. The exact `P-Req-*` queue comes from `scripts/okstra_ctl/plan_items.py` in both branches.
|
|
207
211
|
- **Legacy compatibility details:** assign `R-001`, `R-002`, ... in source order. `Source` uses the existing `brief:` / `derived:` / `contract:` grammar. A `covered` row names the specific Option Candidate and Stage/Step. A gap, blocked clarification, or unaccepted deviation keeps the gate non-passing. `validators/validate-run.py` retains `_validate_requirement_coverage_covered_by`, `_validate_requirement_deviations`, `_validate_gate_blocked_by`, and `_independent_coverage_blockers` enforcement for this branch.
|
|
208
212
|
- **Review-rule compliance plan:** when a project-local review rule pack is found, the chosen realization MUST include the design implication of those rules in its File Structure / interfaces / blast-radius notes. For any helper or data transform used by more than one changed service, the plan must either place it in a shared module or explicitly justify why duplication is intentional. For any test step, the plan must state the observable behavior being asserted, not the internal collaborator call being pinned. For any exported/public method added or renamed, the step must carry the intended noun/side-effect semantics so implementation names can be reviewed before code is written.
|
|
209
|
-
- the
|
|
213
|
+
- the report record MUST include `frontmatter.approved: false` (report-writer always emits the unflipped value). The user authorises the next `implementation` run with `--approve` or the in-session wizard. Do NOT recreate any `User Approval Request` body block — the validator fails reports that contain one (see `validators/validate-run.py` deprecated patterns).
|
|
210
214
|
- Selected-direction plans omit `implementation-option:` because `selectedDirectionRef` already fixes the direction; the legacy-only selector rule is owned by the legacy deliverable section above.
|
|
211
215
|
- **the frontmatter `approved: false` line is rendered unconditionally; if the plan-body verification gate (§5.5.9) returns `blocked-by-disagreement` or `aborted-non-result`, the writer MUST keep `approved: false` and the validator refuses any report that ships with `approved: true` under such a gate result.**
|
|
212
216
|
- every ambiguity flagged during pre-planning that the user must resolve before approval registered as a `Blocks=approval` row in the `## 1. Clarification Items` table (the unified table is the single home for these — the "no separate `Open Questions` block" rule is in the shared `_common-contract.md` clarification policy)
|
|
@@ -40,11 +40,11 @@ roles:
|
|
|
40
40
|
{{INCLUDE:_common-contract.md}}
|
|
41
41
|
{{INCLUDE:_stage-discipline.md}}
|
|
42
42
|
- Pre-implementation gate (mandatory — refuse to start if any item fails):
|
|
43
|
-
- the run brief MUST cite `--approved-plan <path>` pointing to a `final-report.
|
|
44
|
-
- that
|
|
45
|
-
- The `--approve` flag is meaningful ONLY with `--task-type implementation` and `--approved-plan <path>`; any other use raises `PrepareError`. Idempotent — re-running with `approved: true` already set
|
|
43
|
+
- the run brief MUST cite `--approved-plan <path>` pointing to a `final-report-implementation-planning-<seq>.data.json` report record produced by a prior `implementation-planning` run located under `runs/implementation-planning/.../reports/`
|
|
44
|
+
- that plan's report record MUST carry `frontmatter.approved: true`. report-writer emits `false` by default; the user authorises this run with `--approve` or the in-session wizard. Free-form approvals such as "lgtm" / "go ahead" / paraphrased confirmations are NOT accepted; editing the full reading copy does not approve the plan (`okstra_ctl.run._apply_cli_approval`).
|
|
45
|
+
- The `--approve` flag is meaningful ONLY with `--task-type implementation` and `--approved-plan <path>`; any other use raises `PrepareError`. Idempotent — re-running with `approved: true` already set does not write again.
|
|
46
46
|
- determine the plan branch from the sibling data.json `implementationPlanning.planningContract`. For `selected-direction`, the authoritative scope is `selectedDirectionRef`, its validated snapshot, `directionRealization`, and the selected stage; the plan MUST be `plan-ready` with exact coverage, and both an `implementation-option:` frontmatter field and `--implementation-option` are forbidden. A direction change routes to `implementation-option-selection`; a detail-only plan correction routes to `implementation-planning`.
|
|
47
|
-
- for the legacy candidate-comparison branch, the authoritative scope is the Option Candidate named by the
|
|
47
|
+
- for the legacy candidate-comparison branch, the authoritative scope is the Option Candidate named by the report record `frontmatter.implementationOption` field. **If that field is empty, fall back to the plan's `Recommended Option`** (this is a soft fallback, not a hard block). The chosen option's step list becomes the authoritative scope. Any deviation MUST be justified in the final report AND routed to a new `implementation-planning` run; never silently expand scope. If the chosen option name does not match any heading under `Option Candidates`, record it as a deviation.
|
|
48
48
|
- Stage worktree (provisioned by `okstra-ctl` at this implementation run's prep time):
|
|
49
49
|
- Status: `{{EXECUTOR_WORKTREE_STATUS}}` (one of: `created` | `reused` | `skipped-in-worktree` | `skipped-not-git`)
|
|
50
50
|
- Working tree path: `{{EXECUTOR_WORKTREE_PATH}}` — when status is `created` or `reused`, this is this run's isolated stage worktree rooted at `~/.okstra/worktrees/<project>/<task-group>/<task-id>/stage-<N>/`. When skipped, this is the caller's `project_root`.
|
|
@@ -70,13 +70,13 @@ roles:
|
|
|
70
70
|
- uncertainty and overlap relationships kept explicit for downstream consensus classification
|
|
71
71
|
- every candidate's `Expected behavior after` is checked against the brief's `## Preserved Behavior` items before the row is written. A candidate whose observable change contradicts a `PB-NNN` item is dropped, or raised as a `## 1. Clarification Items` row — never softened into a vaguer cell. The scan brief's PB set is the upper bound on what this phase may propose changing.
|
|
72
72
|
- Report assembly instructions:
|
|
73
|
-
- current branch — `schemaVersion: 2.0`: author the structured data contract below and let the independent renderers produce
|
|
73
|
+
- current branch — `schemaVersion: 2.0`: author the structured data contract below and let the independent renderers produce the full reading copy and task-specific human HTML.
|
|
74
74
|
- v1 legacy branch: when validating or rerendering an existing schema-v1 report, preserve its `## 5.9 Improvement Candidates` table and legacy Markdown contract; do not rewrite that historical data into v2 implicitly.
|
|
75
75
|
- the `## 5.9 Improvement Candidates` table populated with rows that obey the 11-column schema from `validators/validate_improvement_report.py` (Cand ID `I-NNN`, Lens from whitelist, Title, Scope ⊆ scan-scope, Severity, Effort, Consensus, Source workers `<worker>:<id>` from {claude, codex, antigravity}, Recommended next-phase ∈ {requirements-discovery, implementation-option-selection, error-analysis}, Expected behavior after, Evidence as path:line list). `Expected behavior after` states, in one observable sentence, what becomes different once the candidate is applied — it is the seed of the downstream brief's `EB-NNN` / `EO-NNN`. A candidate you cannot write this cell for is a preference, not a finding: drop it rather than filling the cell with a restatement of the title.
|
|
76
76
|
- `Consensus` cells in `## 5.9 Improvement Candidates` use the table enum exactly: `full`, `partial`, `contested`, `worker-unique`. Map convergence's `full-consensus` / `partial-consensus` labels to `full` / `partial` before writing the table.
|
|
77
77
|
- Verdict Token — **branch-specific, and the two branches do not share a vocabulary.** On the current v2 branch use the shared analysis enum: `analysis-complete` when every resolved lens was examined, `analysis-partial` when one could not be, `blocked` when the scan itself could not run. `schemas/final-report-v2.0.schema.json` admits only those three for `finalVerdict.verdictToken` — the report's single verdict-token home — so a v2 report carrying `candidates-ready` fails Phase 7. **Finding no candidates is not a verdict**: it is an empty `candidates[]` plus a `lensCoverage[]` row per lens with `status: no-candidate` and its evidence-backed rationale — the verdict stays `analysis-complete`. `candidates-ready` / `no-candidates` belong to the v1 legacy `## 7. Final Verdict` Markdown alone, where `validators/validate_improvement_report.py` enforces them. Both branches: Direction `routing`; Next Step "ask the user to select K candidates (see the ## 5.9 table)".
|
|
78
78
|
- `## 3. Recommended Next Steps` first entry summarises per-candidate routing and proposes new task-key names of the form `<task-group>/imp-<Cand-ID>`
|
|
79
|
-
- author the shared schema-v2 report fields plus `improvementDiscovery.candidates[]`, `improvementDiscovery.lensCoverage[]`, `improvementDiscovery.selectionLimit`, and `improvementDiscovery.userNarrative` in data.json. `candidates[]` carries the same 11 logical fields described above; `lensCoverage[]` records either candidate IDs or an evidence-backed no-candidate rationale for every resolved lens. `schemas/final-report-v2.0.schema.json` and `validators/validate_improvement_report.py` enforce this contract. The renderers independently derive
|
|
79
|
+
- author the shared schema-v2 report fields plus `improvementDiscovery.candidates[]`, `improvementDiscovery.lensCoverage[]`, `improvementDiscovery.selectionLimit`, and `improvementDiscovery.userNarrative` in data.json. `candidates[]` carries the same 11 logical fields described above; `lensCoverage[]` records either candidate IDs or an evidence-backed no-candidate rationale for every resolved lens. `schemas/final-report-v2.0.schema.json` and `validators/validate_improvement_report.py` enforce this contract. The renderers independently derive the full reading copy and human HTML; never author a free-form report.
|
|
80
80
|
- Clarification request policy (phase-specific addenda — shared policy is in `_common-contract.md`):
|
|
81
81
|
- if scan-scope or priority-lenses cannot be made concrete during Phase 1.5, end the run with Verdict Token `blocked`, populate `## 1. Clarification Items` with `Blocks=next-phase` rows, and do not run worker dispatch
|
|
82
82
|
{{INCLUDE:_clarification-recommendation.md}}
|
|
@@ -10,11 +10,11 @@ from okstra_ctl.domain.provider import (
|
|
|
10
10
|
)
|
|
11
11
|
from okstra_ctl.domain.role import ALL_ROLE_TOKENS
|
|
12
12
|
from okstra_ctl.domain.worker_exec import (
|
|
13
|
-
STREAM_JSON,
|
|
14
13
|
ExecCommand,
|
|
15
14
|
PolicySupport,
|
|
16
15
|
WorkerExecRequest,
|
|
17
16
|
)
|
|
17
|
+
from okstra_ctl.domain.worker_presentation import JsonEvents
|
|
18
18
|
from okstra_ctl.domain.worker_stream import (
|
|
19
19
|
Result,
|
|
20
20
|
StreamEvent,
|
|
@@ -224,10 +224,11 @@ class AntigravityExecution:
|
|
|
224
224
|
return ExecCommand(
|
|
225
225
|
argv=tuple(argv),
|
|
226
226
|
stdin_text=None,
|
|
227
|
-
stream_format=STREAM_JSON,
|
|
228
|
-
normalise=step_update_events,
|
|
229
|
-
observe_served_model=observe_served_model,
|
|
230
227
|
cwd=request.project_root,
|
|
228
|
+
presentation=JsonEvents(
|
|
229
|
+
normalise=step_update_events,
|
|
230
|
+
observe=observe_served_model,
|
|
231
|
+
),
|
|
231
232
|
)
|
|
232
233
|
|
|
233
234
|
def policy_support(self) -> PolicySupport:
|
|
@@ -8,11 +8,11 @@ from okstra_ctl.domain.provider import (
|
|
|
8
8
|
)
|
|
9
9
|
from okstra_ctl.domain.role import ALL_ROLE_TOKENS
|
|
10
10
|
from okstra_ctl.domain.worker_exec import (
|
|
11
|
-
STREAM_JSON,
|
|
12
11
|
ExecCommand,
|
|
13
12
|
PolicySupport,
|
|
14
13
|
WorkerExecRequest,
|
|
15
14
|
)
|
|
15
|
+
from okstra_ctl.domain.worker_presentation import JsonEvents
|
|
16
16
|
from okstra_ctl.domain.worker_stream import content_block_events
|
|
17
17
|
|
|
18
18
|
|
|
@@ -129,10 +129,11 @@ class ClaudeExecution:
|
|
|
129
129
|
return ExecCommand(
|
|
130
130
|
argv=tuple(argv),
|
|
131
131
|
stdin_text=request.prompt_text,
|
|
132
|
-
stream_format=STREAM_JSON,
|
|
133
|
-
normalise=content_block_events,
|
|
134
|
-
observe_served_model=observe_served_model,
|
|
135
132
|
cwd=request.project_root,
|
|
133
|
+
presentation=JsonEvents(
|
|
134
|
+
normalise=content_block_events,
|
|
135
|
+
observe=observe_served_model,
|
|
136
|
+
),
|
|
136
137
|
)
|
|
137
138
|
|
|
138
139
|
def policy_support(self) -> PolicySupport:
|
|
@@ -8,11 +8,11 @@ from okstra_ctl.domain.provider import (
|
|
|
8
8
|
)
|
|
9
9
|
from okstra_ctl.domain.role import ALL_ROLE_TOKENS
|
|
10
10
|
from okstra_ctl.domain.worker_exec import (
|
|
11
|
-
TEXT,
|
|
12
11
|
ExecCommand,
|
|
13
12
|
PolicySupport,
|
|
14
13
|
WorkerExecRequest,
|
|
15
14
|
)
|
|
15
|
+
from okstra_ctl.domain.worker_presentation import SplitText
|
|
16
16
|
import okstra_ctl.model_discovery as model_discovery
|
|
17
17
|
|
|
18
18
|
|
|
@@ -85,8 +85,8 @@ class CodexExecution:
|
|
|
85
85
|
return ExecCommand(
|
|
86
86
|
argv=tuple(argv),
|
|
87
87
|
stdin_text=request.prompt_text,
|
|
88
|
-
stream_format=TEXT,
|
|
89
88
|
cwd=request.project_root,
|
|
89
|
+
presentation=SplitText(),
|
|
90
90
|
)
|
|
91
91
|
|
|
92
92
|
def policy_support(self) -> PolicySupport:
|