okstra 0.179.2 → 0.180.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli-registry.mjs +14 -0
- package/dist/cli-registry.mjs.map +1 -1
- package/dist/commands/execute/incremental-carry.mjs +9 -8
- package/dist/commands/execute/incremental-carry.mjs.map +1 -1
- package/dist/commands/execute/plan-verify.mjs +3 -1
- package/dist/commands/execute/plan-verify.mjs.map +1 -1
- package/dist/commands/report/approval-decision.d.mts +1 -0
- package/dist/commands/report/approval-decision.mjs +21 -0
- package/dist/commands/report/approval-decision.mjs.map +1 -0
- package/dist/commands/report/design-snapshot.d.mts +1 -0
- package/dist/commands/report/design-snapshot.mjs +19 -0
- package/dist/commands/report/design-snapshot.mjs.map +1 -0
- package/docs/architecture/storage-model.md +1 -1
- package/docs/architecture.md +10 -10
- package/docs/cli.md +11 -8
- package/docs/project-structure-overview.md +15 -6
- package/docs/task-process/implementation-planning.md +2 -2
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +15 -164
- package/runtime/prompts/launch.template.md +6 -5
- package/runtime/prompts/lead/adapters/cmux.md +1 -1
- package/runtime/prompts/lead/convergence.md +2 -2
- package/runtime/prompts/lead/okstra-lead-contract.md +19 -18
- package/runtime/prompts/lead/plan-body-verification.md +39 -18
- package/runtime/prompts/lead/report-writer.md +64 -423
- package/runtime/prompts/lead/team-contract.md +1 -1
- package/runtime/prompts/profiles/_clarification-recommendation.md +5 -4
- package/runtime/prompts/profiles/_common-contract.md +3 -3
- package/runtime/prompts/profiles/_implementation-deliverable.md +1 -1
- package/runtime/prompts/profiles/change-impact-analysis.md +1 -1
- package/runtime/prompts/profiles/error-analysis.md +1 -1
- package/runtime/prompts/profiles/feature-analysis.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +13 -11
- package/runtime/prompts/profiles/improvement-discovery.md +1 -1
- package/runtime/prompts/profiles/project-analysis.md +1 -1
- package/runtime/prompts/profiles/requirements-discovery.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +2 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
- package/runtime/python/okstra_ctl/agent_activity.py +23 -3
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +6 -6
- package/runtime/python/okstra_ctl/analysis_packet.py +43 -2
- package/runtime/python/okstra_ctl/approval_decisions.py +327 -0
- package/runtime/python/okstra_ctl/design_snapshot.py +134 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +62 -4
- package/runtime/python/okstra_ctl/dispatch_state.py +29 -4
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +6 -2
- package/runtime/python/okstra_ctl/final_report_schema.py +24 -15
- package/runtime/python/okstra_ctl/incremental_carry.py +128 -16
- package/runtime/python/okstra_ctl/incremental_scope.py +4 -1
- package/runtime/python/okstra_ctl/path_hints.py +12 -0
- package/runtime/python/okstra_ctl/paths.py +12 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +113 -16
- package/runtime/python/okstra_ctl/ports/worker_dispatch.py +2 -1
- package/runtime/python/okstra_ctl/render.py +48 -1
- package/runtime/python/okstra_ctl/render_final_report.py +7 -6
- package/runtime/python/okstra_ctl/report_assembly.py +354 -0
- package/runtime/python/okstra_ctl/report_contract.py +2 -1
- package/runtime/python/okstra_ctl/report_finalize.py +60 -22
- package/runtime/python/okstra_ctl/report_inputs.py +72 -0
- package/runtime/python/okstra_ctl/report_markdown.py +69 -8
- package/runtime/python/okstra_ctl/report_narrative.py +319 -0
- package/runtime/python/okstra_ctl/report_projections.py +265 -0
- package/runtime/python/okstra_ctl/run.py +25 -9
- package/runtime/python/okstra_ctl/schema_excerpt.py +11 -6
- package/runtime/python/okstra_ctl/stage_fix_carry.py +4 -4
- package/runtime/python/okstra_ctl/stage_ledger.py +132 -18
- package/runtime/python/okstra_ctl/stage_map.py +70 -22
- package/runtime/python/okstra_ctl/team.py +1 -1
- package/runtime/python/okstra_ctl/worker_dispatch.py +5 -2
- package/runtime/python/okstra_ctl/worker_prompt_body.py +35 -0
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +31 -3
- package/runtime/schemas/final-report-v3.0.schema.json +10210 -0
- package/runtime/schemas/report-narrative-v3.0.schema.json +30 -0
- package/runtime/templates/report-writer-prompt-preamble.md +15 -21
- package/runtime/templates/reports/html/macros/forms.html +6 -4
- package/runtime/validators/validate-run.py +258 -10
|
@@ -55,7 +55,7 @@ Only workers selected from `recommendedWorkers` in `task-manifest.json` and `res
|
|
|
55
55
|
0. **Adapter-owned dispatch (BLOCKING).** Every worker start, await, retry, and shutdown goes through the selected runtime adapter. Core state records the outcome but never guesses a host primitive.
|
|
56
56
|
1. The lead is responsible for orchestration, convergence supervision, and final-report review/approval. It never overrides worker analysis and never bypasses a rostered Report writer worker.
|
|
57
57
|
2. `Report writer worker` is NOT an analysis worker. It is excluded from Phase 4/5 (initial analysis) and Phase 5.5 (convergence re-verification). It is spawned only in Phase 6 and is the **author** of the report record at `runs/<task-type>/reports/final-report-<task-type>-<seq>.data.json`.
|
|
58
|
-
3. When `Report writer worker` is in the roster, Lead MUST dispatch it in Phase 6 as a separate invocation after convergence. Omit it from Phase 4/5 analysis selection and pass `--workers report-writer` for a CLI-backed Phase 6 call.
|
|
58
|
+
3. When `Report writer worker` is in the roster, Lead MUST dispatch it in Phase 6 as a separate invocation after convergence. Omit it from Phase 4/5 analysis selection and pass `--workers report-writer` for a CLI-backed Phase 6 call. Contract v3 has no lead-authored fallback: an attempted dispatch ending in `error` / `timeout` / `not-run` is retried or leaves the run blocked. **Enforced:** `dispatch_core._validate_report_writer_isolation()` rejects every mixed analysis/report plan before process creation, and the default roster selectors exclude `report-writer`; report-writer write paths exclude the final record.
|
|
59
59
|
4. The assigned model for each role is maintained based on `resultContract.requiredWorkerRoles` in task-manifest.json and the lead model metadata.
|
|
60
60
|
5. Required roles must not be replaced by unnamed generic parallel workers.
|
|
61
61
|
6. Before dispatching any required worker, persist the exact worker prompt to the assigned current-run prompt history path under `runs/<task-type>/prompts/`.
|
|
@@ -1,13 +1,14 @@
|
|
|
1
|
-
- every `Kind=decision` clarification row
|
|
1
|
+
- every `Kind=decision` clarification row is recorded by the lead in the approval decision ledger. Each option is an object with eight fields:
|
|
2
2
|
- `role` — `recommended` for the single best answer, `alternative` for the rest. Exactly one option per row is `recommended`.
|
|
3
3
|
- `answer` — the choice itself, phrased so the user can pick it as-is. Keep it to a short phrase (roughly 120 characters); the reasoning and the consequences have their own fields below.
|
|
4
4
|
- `rationale` — one sentence on why this option is on the board.
|
|
5
|
-
- `
|
|
5
|
+
- `reach` — exactly one of `in-repo` or `cross-repo`.
|
|
6
|
+
- `scopeEffects` — optional tokens drawn from `{new-schema, deferrable}`.
|
|
6
7
|
- `addedWork` — one sentence naming the work this choice creates that the other choices do not. Name the work, not a cost adjective.
|
|
7
8
|
- `directionChange` — one sentence naming what this choice reverses: an approved plan item, a recorded decision, an earlier answer. When it reverses nothing, say so.
|
|
8
9
|
- `disposition` — the effect of selecting the option. Use `select` for `user-decision`, `accept-risk` for `noncritical-dissent`, and `request-revision` or `reject` when the option sends the plan back. `correctness-critical` never offers `accept-risk`.
|
|
9
|
-
-
|
|
10
|
+
- report assembly derives `approvalContext`, status, and resolution. `approvalContext` contains only `classification`, `unblockCondition`, and `recommendedDisposition`; it never copies plan or activity identifiers.
|
|
10
11
|
- the three impact fields answer three different questions — how far the change reaches, what new work it creates, and what it overturns. Someone choosing between options needs all three, so never fold them into one sentence: whichever axis is easiest to write would silently stand in for the other two.
|
|
11
12
|
- a row that omits `options[]`, offers fewer than two, or marks zero or two options as `recommended` is incomplete and must be completed before the report is finalised.
|
|
12
13
|
- `expectedForm` states only the *shape* of the answer — one of the options, a file path, a number, a date. It never lists the choices again; two sources for one fact leave consumers disagreeing about which is authoritative.
|
|
13
|
-
- **Enforced:** `
|
|
14
|
+
- **Enforced:** `scripts/okstra_ctl/approval_decisions.py`, `schemas/final-report-v3.0.schema.json`, and `scripts/okstra_ctl/report_assembly.py`.
|
|
@@ -14,10 +14,10 @@ profile document.
|
|
|
14
14
|
- For a new `implementation-planning` run, the plan-body sequence is initial verification → one planner self-fix → targeted re-verification → user gate. The initial verification is round 1, the targeted re-verification is round 2, and a second automatic self-fix is a contract violation. A user-directed correction does not consume the automatic self-fix limit, and a verification failure after that correction does not restart the automatic loop.
|
|
15
15
|
- **provider-unavailable fallback (tolerance).** A worker dispatch can fail to produce a result for two distinct reasons, and both take the same recovery path. (1) **Pane budget:** the dispatch is rejected because a teammate pane could not be created — this is the harness running out of room for its own teammate panes, not okstra placing a worker. The wording is the host's, so match the condition rather than a fixed string. (2) **Sandbox CLI-start failure:** an external CLI worker wrapper exits non-zero within seconds with empty stdout and its live-log shows `operation not permitted`. In either case the lead spends the one shared retry budget through the assignment's recorded runner. If the provider is still unavailable, record that terminal status and continue only under the convergence quorum rules; never replace it silently with a fixed provider or count a substitute as the original provider's vote. Completed external-CLI workers hold no pane of their own. A pane the harness opened for its own teammate carries no id okstra recorded, so no okstra command closes it — the host and the user own that surface. (This is a prompt instruction, not a code-enforced gate.)
|
|
16
16
|
- Dual-audience final-report contract (shared):
|
|
17
|
-
-
|
|
17
|
+
- The report writer authors the report narrative Markdown. Report assembly combines it with role-owned machine inputs and publishes data.json once; the reading copy and human HTML are derived from that record.
|
|
18
18
|
- User-facing information belongs in `humanSummary` and the selected task block's `userNarrative`; it must not exist only in Markdown. The HTML human main body explains the result with those fields plus task facts.
|
|
19
19
|
- Agent coordination and audit details belong in `crossVerification`, `executionStatus`, and `tokenUsage`. HTML may expose them only inside collapsed audit details, never as the primary result.
|
|
20
|
-
- Clarification and approval controls are rendered from
|
|
20
|
+
- Clarification and approval controls are rendered from the lead-owned approval decision ledger after report assembly. A question that could have been resolved before dispatch through the profile or Reporter Confirmations is an intake failure, not a final-report question.
|
|
21
21
|
- Tooling — read-only MCP availability (shared):
|
|
22
22
|
- MCP is not implicit context; query a server only when the task brief explicitly lists it as source material for this run. Any MCP-derived finding MUST cite server, table, and the SELECT used. MCP MUST NEVER be a write path — schema/data mutations go through repository migration files reviewed by humans.
|
|
23
23
|
- Resource boundary (shared — artifact-home rule):
|
|
@@ -81,7 +81,7 @@ profile document.
|
|
|
81
81
|
Profile-specific addenda may tighten cell content but MUST NOT add, remove, rename, or reorder columns, nor change the meta-cell field order. The `ID` is `C-NNN` (3-digit zero-padded), the `Status` ∈ `{open, answered, resolved, obsolete}`, and the `Kind` / `Blocks` legal values are listed below.
|
|
82
82
|
- In schema-v1 Markdown and worker-result tables, section 1 is a **single unified table** per `final-report-template.md`. Every clarification item is one row. Do not split it into sub-sections or create a parallel question table.
|
|
83
83
|
- each row's `Kind` column picks one of `{material, decision, data-point}`: `material` for files / snapshots / logs / screenshots the user must attach (the `User input` cell will hold a path or URL); `decision` for choices and yes/no confirmations only the user can make; `data-point` for a single number, ID, date, or short string the user can answer inline. A `decision` alternative must be a terminal choice the user can pick as-is; if acting on an alternative still requires the user to supply a concrete value (a path, string, number, or file), that value is its own `data-point` / `material` row — never phrase a data-entry action (e.g. "specify the path", "enter a value") as a selectable `decision` option, because the rendered `<select>` cannot capture the value the option demands. Items that mix "yes/no + file path if yes" are one row of `Kind=material` with the combined expectation written into `Expected form`.
|
|
84
|
-
- **One decision per row.** A `decision` row asks one question. When a single option bundles two independent decisions
|
|
84
|
+
- **One decision per row.** A `decision` row asks one question. When a single option bundles two independent decisions, split the row. The tell is usually the option's `reach`: an option that is `cross-repo` only because one bundled clause crosses a repository boundary contains two decisions of different cost. The §5.5.9 adversarial round judges this semantic rule.
|
|
85
85
|
- each row's `Blocks` column picks one of `{approval, next-phase, none}`. `approval` is reserved for items that gate an approval action, especially the `implementation-planning` `approved:` frontmatter flip; outside `implementation-planning`, unresolved brief reporter-confirmation rows use `next-phase` instead. `next-phase` blocks the next run from starting cleanly. `none` is informational/audit-only.
|
|
86
86
|
- write every entry in full, descriptive sentences that a non-developer can act on without further context. Avoid abbreviations and internal jargon. The `Statement` cell must state *what* is needed, *why* the answer / attachment changes the next step, and (for `material`) *where* the user can find it and *where* to place it. The `Expected form` cell must state the answer shape (yes/no, one of the options, number/date, file path, short description, etc.); supply concrete option choices when applicable.
|
|
87
87
|
- **Record coordinates only.** A clarification `statement`, `expectedForm`, or `options[]` answer/rationale may cite a report-record row id (`RB-002`, `C-014`) or a `path:line`. Do not cite a section number (`§4.7`, `§1`). That number exists only on one full reading copy. **Enforced:** `validators/validate-run.py` `_validate_clarification_record_coordinates`.
|
|
@@ -52,7 +52,7 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
52
52
|
git diff <base>..HEAD | grep -E '^\+[^+].*\b(TBD|TODO|FIXME|XXX|implement later|handle edge cases|similar to|placeholder)\b' || echo 'clean'
|
|
53
53
|
```
|
|
54
54
|
Only newly-added lines (those starting with `+` and not part of the `+++` header) are inspected. If output is anything other than `clean`, the run MUST either remove the placeholders before finalising or record an explicit justification per occurrence in the final report.
|
|
55
|
-
7. **Stage-foreign literal scrub** — when the report-writer
|
|
55
|
+
7. **Stage-foreign literal scrub** — when the report-writer models this stage's narrative on another stage's report, stage-specific literals can be copied verbatim and misattribute this run. Confirm every branch name, commit SHA, stageKey, and stage number in the narrative resolves to **this** run's stage `<N>` — its worktree branch is `<prefix>-<task-id>-s<N>`, its stageKey `<task-id>-stage-<N>`. Sweep the narrative for any `-s<M>` / `stage-<M>` / `Stage <M>` where `M ≠ N` and for SHAs not in this run's `Commit list`; each hit is a copy-from-other-stage defect to correct before finalising.
|
|
56
56
|
8. **Manual user test coverage** — when this run changed user-observable behaviour, §5.7.9 must carry concrete steps (not the `applicable=false` exemption). A user-facing change shipped with an exemption line is a contract violation; an internal-only change with no observable surface is the only valid use of the exemption.
|
|
57
57
|
|
|
58
58
|
## Lead post-stage persistence (BLOCKING — runs after the Executor emits `### Stage Carry Evidence`)
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
roles:
|
|
5
5
|
- role: planner
|
|
6
6
|
min: 2
|
|
7
|
-
recommended:
|
|
7
|
+
recommended: 3
|
|
8
8
|
max: 5
|
|
9
9
|
duty: planning-worker
|
|
10
10
|
- role: critic
|
|
@@ -66,7 +66,9 @@ roles:
|
|
|
66
66
|
- **Stage Ledger (read before drafting the Stage Map):** when this task already has a plan on disk, the analysis packet carries a `## Stage Ledger` JSON block listing every stage with its `status` (`done` / `active` / `ready` / `blocked`), `dependsOn`, and done commit. It states what exists, not what to plan. Two rules follow from it:
|
|
67
67
|
- A stage whose `status` is `done` is already implemented and will not be executed again. Carry its plan body forward as written; do not rewrite its steps, and do not fold its work into a new stage.
|
|
68
68
|
- Every stage number in the ledger is taken. A new stage takes the next number after the highest one listed; numbers are never reused or reordered. **Not yet machine-enforced** — the validator for this rule lands with the plan-amendment feature.
|
|
69
|
-
- The block is
|
|
69
|
+
- The ledger answers two questions from two sources, and the block names both. `sourcePlan` is the plan the completed stages were actually built against; `latestPlan` is the plan the `stages` list came from and is therefore the numbering authority. When they differ, the completed work followed the former and the highest taken number comes from the latter.
|
|
70
|
+
- A `planDivergence` entry means the two plans disagree about a stage that is already `done` — the same number naming different work, or a completed stage the latest plan no longer declares. Do not pick one of the two yourself; register it as a `Blocks=approval` clarification row and assign no new stage number until it is resolved.
|
|
71
|
+
- The block is absent ONLY on a task's first planning run. Its absence then means there is no prior plan, not that no stage is done. When the ledger could not be read, the packet says so under the same heading with a reason instead of going silent — in that state, assign no new stage number and report the reason as a blocker.
|
|
70
72
|
- Primary focus areas:
|
|
71
73
|
- requirement gaps
|
|
72
74
|
- affected components and boundaries
|
|
@@ -89,12 +91,12 @@ roles:
|
|
|
89
91
|
- Implementation Design Preparation (`implementation-design-prep-v1`, BLOCKING):
|
|
90
92
|
- **Detector SSOT:** the planner MUST run the V1 detector defined by `scripts/okstra_ctl/design_surfaces.py` (`detect_design_surfaces()` over the detector's `RULES`) and MUST NOT invent or copy a second keyword list into the plan or prompt. **Enforced:** `validators/validate-run.py` `_validate_detector_coverage` reruns that detector and compares every `(stage, kind)` plus its trigger evidence.
|
|
91
93
|
- **Exactly-once coverage:** for every detector-produced `(stage, kind)`, the planner MUST write exactly one `designSurfaceCoverage` row on that stage. **Enforced:** `validators/validate-run.py` `_validate_detector_coverage` rejects missing, duplicate, extra-detector-kind, or evidence-mismatched rows; `schemas/final-report-v2.0.schema.json` `$defs.DesignSurfaceCoverage` enforces the row shape.
|
|
92
|
-
- **Disposition:**
|
|
94
|
+
- **Disposition:** the design-surface detector snapshot uses `inline-contract` only when the stage already states the kind-specific minimum implementation contract; otherwise it uses `prep-item`. `not-applicable` is legal only with a concrete rationale consistent with the stage action. **Enforced:** `scripts/okstra_ctl/report_projections.py::project_design`, `schemas/final-report-v3.0.schema.json`, and `P-Prep-S<stage>-<kind>` verification.
|
|
93
95
|
- **AI-prepared proposal:** every referenced PREP item MUST record `kind`, `stageRefs`, `need`, evidence-cited `knownFacts`, `openQuestions`, a concrete evidence-backed `aiProposal` (`summary`, `details`, `assumptions`, `evidence`, `confidence`), `humanConfirmation`, explicit `status`, and the safest reversible default available. **Enforced:** `schemas/final-report-v2.0.schema.json` `$defs.DesignPrepItem` / `$defs.DesignPrepProposal` enforce required fields, `validators/validate-run.py` `_validate_prep_references` enforces the bidirectional stage/kind link, and `prompts/lead/plan-body-verification.md` rejects empty or non-implementable proposals.
|
|
94
96
|
- **Status choice:** prefer `provisional` with a `workingAssumption`, concrete `guardrails`, `reviewAt`, `ifStillOpen`, and canonical `requestPath`; use `blocked` only for business policy, external authority, a destructive migration decision, or the absence of any safe reversible assumption. **Enforced:** `schemas/final-report-v2.0.schema.json` `$defs.DesignPrepItem` enforces state-specific fields, `validators/validate-run.py` `_validate_design_prep_states` enforces confirmation/request invariants, and `prompts/lead/plan-body-verification.md` judges whether the disposition is justified. A declared `blocked` status does not by itself fail plan-body verification.
|
|
95
97
|
- **External-reality anchoring (`external-interface` / `transformation-mapping` surfaces):** these two detector kinds are correct only against data whose shape lives *outside this repository* (a third-party response body / external payload format). For such a surface the referenced PREP item's `aiProposal` MUST derive the assumed shape — selectors, field paths, response structure — from a **captured real sample** and cite it in `knownFacts` / `evidence` (source + capture time); a shape invented from internal reasoning and marked `confidence: high` is the disallowed move, because a plan built on an assumed shape yields an implementation whose parser and fixture only ever agree with each other. When the brief supplies no sample and none is capturable at plan time, the item MUST stay `provisional` with a `workingAssumption` that the external shape is unverified against reality, a `guardrails` line forbidding the implementation from presenting a synthetic-fixture green run as reality-verified, and an `ifStillOpen` that routes to user confirmation against real data — it MUST NOT be dispositioned `inline-contract` / settled. **Enforced (semantic):** the surface's presence is machine-checked by `_validate_detector_coverage`; whether its proposal's evidence is genuinely external is judged by the §5.5.9 `P-Prep-S<stage>-<kind>` round (this phase runs it adversarially).
|
|
96
98
|
- **Planner-only test surface:** add `manual-user-test` only when a test prerequisite changes the implementation interface or acceptance contract; the V1 detector never emits it. **Enforced:** `validators/validate-run.py` `_validate_detector_coverage` rejects detector-produced `manual-user-test`, `schemas/final-report-v2.0.schema.json` permits its planner-authored shape, and `prompts/lead/plan-body-verification.md` verifies the stage-action rationale.
|
|
97
|
-
- **Trivial task:** when the detector returns no surfaces
|
|
99
|
+
- **Trivial task:** when the detector returns no surfaces, its input uses `designPreparation.mode: no-design-inputs`, an empty `items` array, and a concrete reason tied to the plan. **Enforced:** `schemas/final-report-v3.0.schema.json` and `scripts/okstra_ctl/report_assembly.py`.
|
|
98
100
|
- Approval gate (phase-specific addendum to shared authority rule):
|
|
99
101
|
- The report record `frontmatter.approved` field is the only authorised approval gate. report-writer always emits `false`. The user clears it by invoking the next phase with `--approve`, or by confirming approval in the in-session wizard. Editing the full reading copy does not approve the plan. `okstra_ctl.run._validate_approved_plan` reads this field and refuses entry until it is `true`.
|
|
100
102
|
- Cross-verification mode:
|
|
@@ -102,7 +104,7 @@ roles:
|
|
|
102
104
|
- §5.5.9 plan-body verification runs with an **adversarial posture** (`prompts/lead/plan-body-verification.md` §"Adversarial plan-body posture"): verifiers open and confirm every cited path / command and put the burden of proof on the plan. The gate threshold is majority-based for kinds `b`/`c`/`e`, but a single `DISAGREE` blocks on its own for the concrete, safety-critical kind `a` (path/symbol mismatch) — and `f` on `P-Req-*` items. `P-Var-*` items are excepted from the kind-`a` exception: a variation-point defect takes a majority. Rollback ordering (`d`) is advisory and never blocks the gate — a rollback is executed by a human, not by okstra's workers or verifiers. A majority also needs ≥2 participating votes, so a lone dissent whose peer returned a non-result does not block on a majority-gated kind (see that contract's §"Adversarial plan-body posture").
|
|
103
105
|
- **Incremental re-verification scope (clarification re-runs):** when the lead's `okstra incremental-scope` decision is `mode == "incremental"` (procedure in `prompts/launch.template.md` §"Clarification Response Carried In"), workers re-analyze ONLY the stages listed in `reverify_stages` (the downstream closure of the impacted stages). Workers MUST NOT re-open, re-score, or re-judge any stage in `carry_stages` — those stages' prior plan-item verdicts are carried forward verbatim, and a worker never overwrites a carried verdict with its own judgement. When the decision is `mode == "full"` (the default), every stage is re-analyzed as usual.
|
|
104
106
|
- **Single incremental-scope decision:** the lead calls `okstra incremental-scope` exactly once for the re-run, passing the answered `C-NNN` ids through `--answered-clarifications`, changed design-preparation IDs through `--prep-items`, and any lead-resolved stage numbers through `--impacted`; the CLI unions all three before applying the existing dependency closure and cutoff. The clarification ids are resolved to stages by the CLI from the prior report's own `planItems[].clarificationId` and `blocked C-NNN` coverage links — the lead does not map answers to stage numbers. An answer that changes the selected planning payload, Stage Map, or execution approach is not a local impact: pass every CSV empty so the same call returns `mode == "full"`. A clarification id that traces to no stage, unknown PREP IDs, or invalid `stageRefs` also return an explicit full decision instead of being guessed. When the user pinned a scope at the wizard (`REVERIFY_SCOPE_MODE` / `REVERIFY_SCOPE_STAGES` in `prompts/launch.template.md` §"Clarification Response Carried In" step 0), that pin is an input to this same single call — `full` supplies the `--full-reason`, and pinned stage numbers join `--impacted` — never a second call or a bypass of the CLI's closure and cutoff.
|
|
105
|
-
- **Stage-aware carry:** for an incremental decision,
|
|
107
|
+
- **Stage-aware carry:** for an incremental decision, the report writer copies each `carry_stages` stage row unchanged into its narrative. After plan-item seeding, pass the decision's `carry_stages` and `reverify_stages` CSVs unchanged to `okstra incremental-carry --cur-narrative ... --state ... --out-state ...`. The helper rejects a changed or missing carried stage and copies only its prior `P-Step-*` / `P-Prep-*` verdicts into the convergence-owned state. Overlap, omissions, and canonical conflicts return `CarryError`. On that error, discard the partial state and run full re-verification.
|
|
106
108
|
{{INCLUDE:_coverage-critic.md}}
|
|
107
109
|
- Non-goals:
|
|
108
110
|
- code-level micro-optimization unless it changes the implementation approach
|
|
@@ -129,7 +131,7 @@ roles:
|
|
|
129
131
|
- `tddExemption` waives both rules above, and only for `doc-only`, `config-only`, or `pure-rename` work. An empty or arbitrary reason waives nothing. **Enforced (S10e):** same function — the schema alone cannot reject it, because it types the field as a plain string and keys its conditional on the property merely being present.
|
|
130
132
|
- `stageMap[].dependsOn` must form a DAG (no self-dependency, no unknown stage, no cycle), each row's `stepCount` must equal its stage's actual `stepwiseExecution` row count, and two `(none)`-dependency stages must not name the same file in their `exitContract` — they run as concurrent implementation runs in separate worktrees. **Enforced (S8/S4/S9):** same function.
|
|
131
133
|
- Legacy candidate-comparison-only deliverable:
|
|
132
|
-
- The
|
|
134
|
+
- The report writer records the plan body in report narrative Markdown under `implementationPlanning`. Report assembly adds machine-owned `designPreparation`, `designSurfaceCoverage`, and `planBodyVerification`, then validates the completed `data.json` against `schemas/final-report-v3.0.schema.json`.
|
|
133
135
|
- Legacy candidate-comparison requires at least two implementation options. **Each option must include**:
|
|
134
136
|
- **File Structure**: an explicit list of files to create / modify / delete with each file's responsibility (one-line each). Use the form `Create: path — responsibility` / `Modify: path:line-range — change summary` / `Delete: path — reason`. Write every `path` in full and `<PROJECT_ROOT>`-relative — never ellipsis-abbreviated (`…` / `...` / a trailing `/…`); an abbreviated path does not resolve and is rejected by plan-body verification as a kind-b path mismatch.
|
|
135
137
|
- **Two-tier change description.** Each `fileStructure` row carries `summary` **and** optional `details`, and they are not interchangeable:
|
|
@@ -161,7 +163,7 @@ roles:
|
|
|
161
163
|
**Pick the cases by distinct outcome, not by line coverage.** When the stage writes or reconciles state, its meaningfully different outcomes are usually more than three — normal success, target already in the desired state (resume), existing data reused rather than created, a conflicting concurrent state, target absent, and mid-way failure with rollback. Enumerate the ones this stage actually implements and route them across the three lines (the `boundary` line is where resume / already-done / reuse belongs; `failure` carries conflict, absence, and rollback), naming each in the cell rather than collapsing them into "edge input". An implemented outcome with no declared case is a coverage gap the executor will not backfill.
|
|
162
164
|
- **Per-stage subsections** (`## 5.5.<i> Stage <i>: <title>` for each `i`), each containing the four required subsections:
|
|
163
165
|
- `### Carry-In` — for `depends-on (none)`: task-brief only. Otherwise: each depended-on stage's static exit contract + runtime sidecar path `runs/<impl-key>/carry/stage-<i>.json` placeholder.
|
|
164
|
-
- `### Stepwise Execution Order` — bite-sized table with `step | action | files | command | outcome | expected`. `outcome` is one word — `PASS` or `FAIL` — and `expected` is the sentence saying what that looks like here; a verdict written inside the sentence is not read as one. The `files` cell lists each touched path in full and `<PROJECT_ROOT>`-relative — never ellipsis-abbreviated (`…` / `...`), which does not resolve and is rejected by plan-body verification as a kind-b path mismatch. **The
|
|
166
|
+
- `### Stepwise Execution Order` — bite-sized table with `step | action | files | command | outcome | expected`. `outcome` is one word — `PASS` or `FAIL` — and `expected` is the sentence saying what that looks like here; a verdict written inside the sentence is not read as one. The `files` cell lists each touched path in full and `<PROJECT_ROOT>`-relative — never ellipsis-abbreviated (`…` / `...`), which does not resolve and is rejected by plan-body verification as a kind-b path mismatch. **The narrative row additionally carries `plannedPaths`: the same paths as an array, one repository-relative path per entry, with no globs, exclusions, counts or commentary.** `files` is the sentence a reader sees; `plannedPaths` is the ledger report assembly preserves and the implementer write policy enforces. When a step legitimately covers a set too large to enumerate, split it or name the directory the set lives under. **Effective row count ≤ 8** (excluding header / divider / blank). Each step is one cohesive, self-contained change. **TDD ordering is MUST, not a preference:** the **first** effective step's `action` cell MUST start with the literal `RED:` and describe the failing test(s) that capture this stage's `Acceptance` **and the three declared `Test case (success|boundary|failure)` lines** (`outcome` = `FAIL`); at least one later `action` cell MUST start with the literal `GREEN:` and describe the minimal implementation that makes it pass (`outcome` = `PASS`); an optional refactor step starts with `REFACTOR:`. **Exemption:** doc-only / config-only / pure-rename stages with no observable runtime behaviour may omit RED/GREEN by declaring one line `TDD exemption: <reason>` in the stage section. Validator S10c enforces RED-first + GREEN; the `outcome` cell agreeing with its prefix is a schema conditional. S10e rejects an unsupported exemption reason (`validators/validate-implementation-plan-stages.py`).
|
|
165
167
|
- **The `command` cell runs inside an okstra task worktree, not a bare checkout (BLOCKING).** okstra provisions `.okstra`, the configured sync entries (`.project-docs`, `.claude`, …), and — for `implementation` — a nested `stage-<N>/` worktree into the tree the step executes in. Two consequences bind every command you write:
|
|
166
168
|
- **Clean-tree assertions use `okstra worktree-status --check-clean`.** A bare `git status --porcelain` is never empty there, so an assertion built on one fails on okstra's scaffolding rather than on the stage's work. The okstra command asks the same question over source paths only and exits 1 when dirty, so it stands alone as a step's assertion: `okstra worktree-status --check-clean`. Validator S13 rejects the bare form. Do not add a `git tag stage-<N>-exit` to the step — okstra writes that tag itself when it settles the stage, at the commit the carry evidence records, and a step that tags mid-stage puts it on an earlier commit.
|
|
167
169
|
- **Never read an `.okstra/` artifact back out of a git object.** `.okstra/**` is gitignored and never committed — the executor aborts a commit that stages an ignored path and the verifier reports a committed `.okstra` path as a branch defect — so `git cat-file -e <tag>:.okstra/…`, `git show <tag>:.okstra/…`, and every variant of that read can never resolve, at any tag, in any stage. A later stage that needs a QA artifact reads it from the working tree or receives it through the carry sidecar / verifier result; do not design a stage contract around one being reachable from a tag. Validator S12 rejects the read.
|
|
@@ -214,7 +216,7 @@ roles:
|
|
|
214
216
|
- Selected-direction plans omit `implementation-option:` because `selectedDirectionRef` already fixes the direction; the legacy-only selector rule is owned by the legacy deliverable section above.
|
|
215
217
|
- **the frontmatter `approved: false` line is rendered unconditionally; if the plan-body verification gate (§5.5.9) returns `blocked-by-disagreement` or `aborted-non-result`, the writer MUST keep `approved: false` and the validator refuses any report that ships with `approved: true` under such a gate result.**
|
|
216
218
|
- every ambiguity flagged during pre-planning that the user must resolve before approval registered as a `Blocks=approval` row in the `## 1. Clarification Items` table (the unified table is the single home for these — the "no separate `Open Questions` block" rule is in the shared `_common-contract.md` clarification policy)
|
|
217
|
-
- **Exact plan-item queue (BLOCKING).** Run `okstra plan-items extract --
|
|
219
|
+
- **Exact plan-item queue (BLOCKING).** Run `okstra plan-items extract --narrative <report-narrative.md> --output <state>/plan-items-....json`, place the persisted `items[]` verbatim in the verifier prompt, then run `okstra plan-items validate --narrative <report-narrative.md> --items <state>/plan-items-....json`. Do not freely summarise, select, omit, reorder, or renumber the queue. Prompt headings use the compact `subject` and include the lossless `payload`. For every item, ask:
|
|
218
220
|
|
|
219
221
|
```text
|
|
220
222
|
What concrete false-positive input, failure ordering, or omitted dependency
|
|
@@ -223,13 +225,13 @@ roles:
|
|
|
223
225
|
|
|
224
226
|
An `AGREE` note records the counterexample considered and its exclusion reason. If the judgement needs unavailable external material, record `verification-error`, not `DISAGREE`. **Enforced:** `validators/validate-run.py` `_validate_plan_item_extraction_completeness` compares the exact deterministic set, independently rejecting missing, unexpected, and duplicate plan-item IDs, including `P-Prep-*`.
|
|
225
227
|
- **§5.5.9 Plan Body Verification (BLOCKING).** After report-writer finishes the draft, the lead runs a worker peer-review round on the persisted queue. Selected-direction plans begin with `P-Dir-1`; legacy candidate-comparison plans begin with `P-Opt-*`; both continue with the shared execution items. The fixed order remains initial verification → one planner self-fix → targeted re-verification → user gate. Gate recomputation, extraction completeness, approval-context reconciliation, and self-fix limits remain enforced by `validators/validate-run.py`; verdict details and dissent format are owned by `prompts/lead/plan-body-verification.md`.
|
|
226
|
-
- **Approval decision state.**
|
|
228
|
+
- **Approval decision state.** The lead records active and carried decisions through `okstra approval-decision`. Every option carries `disposition`, one `reach`, and optional `scopeEffects`. A resolved decision names existing `A-NNN` checks; report assembly derives `approvalContext`, status, resolution, and reverse links.
|
|
227
229
|
- `open → answered` when the raw user response is recorded
|
|
228
230
|
- `answered → resolved` only after the selected disposition is applied and its checks pass
|
|
229
231
|
- `answered → open` when application or checking fails
|
|
230
232
|
- `open → obsolete` only when a plan change removes the question
|
|
231
233
|
`open` and `answered` block approval; only `resolved` and `obsolete` are non-blocking. **Enforced:** `validators/validate-run.py` `_validate_approval_context` plus run-prep `scripts/okstra_ctl/run.py` `_validate_approved_plan`.
|
|
232
|
-
- **Terminal approval evidence.** A
|
|
234
|
+
- **Terminal approval evidence.** A resolved correctness-critical decision names a later successful evaluation through `resolutionInput.checkRefs`. Each referenced activity carries the same `C-NNN` in `clarificationRefs[]`, the affected `planItemIds[]`, zero-exit commands, and the plan-body state result. Report assembly rejects a missing activity or reverse link before publication. **Enforced:** `scripts/okstra_ctl/report_assembly.py::_clarification_row` and `_attach_plan_backlinks`.
|
|
233
235
|
- **Decision-record evaluation (sole owner)**: this phase is the **single owner** of decision-record evaluation in the okstra lifecycle. The brief never evaluates or drafts decision records — it only forwards `adr-candidate:*` signals. Every `adr-candidate:*` entry inherited from the brief's `Open Questions` is a mandatory evaluation target. In addition, evaluate every decision the chosen realization introduces against the three criteria:
|
|
234
236
|
1. **Hard to reverse** — would changing the decision later cost meaningfully more than deciding now?
|
|
235
237
|
2. **Surprising without context** — would a future reader, seeing only the code, wonder "why was it built this way?"?
|
|
@@ -254,7 +256,7 @@ roles:
|
|
|
254
256
|
9. **Cross-project dependency check** — confirm you have not missed a dependency on another repo / another top-level deployable module / a published package. If `dependencyMigrationRisk` has a `kind: cross-project` row, confirm a matching `direction: upstream-precondition` `XP-NNN` row exists in `crossProjectDependencies`, and re-read as a reviewer whether its `requiredWork` is the concrete work the other side must actually build rather than an abstract phrase ("other side's work done") — validator S only checks existence, so concreteness is the self-review's responsibility. Confirm cross-repo work is split into a separate run + XP row instead of being crammed into one task's stages, and that the cross-project substance is not duplicated in `§3 Recommended Next Steps` but lives only in `§5.4 Cross-Project Dependencies`.
|
|
255
257
|
10. **Decision-draft materialization check** — when `decisionDrafts` is non-empty, confirm as a reviewer which stage's stepwise order contains the matching materialization step (creating `.okstra/decisions/<NNNN>-<slug>.md`) and that the number of drafts corresponds 1:1 with the materialization steps. The validator only checks the *existence* of the step, so the `<NNNN>-<slug>` correctness and count correspondence are the self-review's responsibility.
|
|
256
258
|
11. **Variation-point & seam check** — read `variationPointAnalysis` as a skeptic. Is `hasMultipleImplementations` honest against the brief and the sibling code you inspected during pre-planning, or was `false` chosen because it is the cheaper field to fill? For every point with `extract: true`, confirm the `extractionDecision` names a real interface (a `port` for a hexagonal project, not a shared helper) and a `coveredBy` stage that exists in the Stage Map — an interface no stage builds is a decision nobody executes. Then read the chosen realization's `testSeams`: each `injectedAs` must name a construction or wiring point a test can actually substitute at, not a symbol the test would have to re-implement — a seam nothing can be injected into leaves the executor writing self-mocks. An empty `testSeams` array is only acceptable when you can defend it in one sentence; the validator accepts it either way, so this is the check that catches an unfilled field posing as a decision.
|
|
257
|
-
12. **Approval blast-radius check (BLOCKING).** Every
|
|
259
|
+
12. **Approval blast-radius check (BLOCKING).** Every approval clarification must be reachable from `planItems[].clarificationRefs[]` or a requirement-coverage blocker. Report assembly derives plan-item links from activity `clarificationRefs[]` plus `planItemIds[]`; `okstra incremental-scope` reads the resulting reverse links.
|
|
258
260
|
- **The link must resolve to a stage, not merely exist.** `incremental-scope` reads the stage number out of a `P-Step-<stage>.<step>` / `P-Prep-S<stage>-<kind>` plan-item id, or out of a `Stage N` citation in the blocked coverage row's `coveredBy`. Every other plan-item prefix (`P-Dir-1`, `P-Req-*`, `P-Val-*`, `P-Opt-*`, `P-Dep-*`, `P-Rb-*`) carries no stage, so a blocker linked only that way MUST also have its coverage row cite the stage in `coveredBy`. Writing the blocked row's `coveredBy` as prose with no `Stage N` in it — `No stage.`, `Partly covered — …` — satisfies nothing: the row passes the link check and the re-run still re-verifies everything.
|
|
259
261
|
- What to write when no stage covers the requirement yet: name the stage the answer will change, not the stage that satisfies the requirement today. A `Blocks=approval` row is admissible only when, absent an answer, `implementation` would produce wrong or unsafe code (see the admissibility rule above) — so some stage's code is at stake by construction. If you genuinely cannot name one, the row fails the admissibility test and belongs in `## 5. Missing Information and Risks` with `Blocks=none`, not in the approval gate.
|
|
260
262
|
**Enforced:** `validators/validate-run.py` `_validate_approval_clarification_backtrace` — one failure for a missing link, a separate one for a link that resolves to no stage.
|
|
@@ -76,7 +76,7 @@ roles:
|
|
|
76
76
|
- `Consensus` cells in `## 5.9 Improvement Candidates` use the table enum exactly: `full`, `partial`, `contested`, `worker-unique`. Map convergence's `full-consensus` / `partial-consensus` labels to `full` / `partial` before writing the table.
|
|
77
77
|
- Verdict Token — **branch-specific, and the two branches do not share a vocabulary.** On the current v2 branch use the shared analysis enum: `analysis-complete` when every resolved lens was examined, `analysis-partial` when one could not be, `blocked` when the scan itself could not run. `schemas/final-report-v2.0.schema.json` admits only those three for `finalVerdict.verdictToken` — the report's single verdict-token home — so a v2 report carrying `candidates-ready` fails Phase 7. **Finding no candidates is not a verdict**: it is an empty `candidates[]` plus a `lensCoverage[]` row per lens with `status: no-candidate` and its evidence-backed rationale — the verdict stays `analysis-complete`. `candidates-ready` / `no-candidates` belong to the v1 legacy `## 7. Final Verdict` Markdown alone, where `validators/validate_improvement_report.py` enforces them. Both branches: Direction `routing`; Next Step "ask the user to select K candidates (see the ## 5.9 table)".
|
|
78
78
|
- `## 3. Recommended Next Steps` first entry summarises per-candidate routing and proposes new task-key names of the form `<task-group>/imp-<Cand-ID>`
|
|
79
|
-
- author the shared
|
|
79
|
+
- author the shared narrative fields plus `improvementDiscovery.candidates[]`, `improvementDiscovery.lensCoverage[]`, `improvementDiscovery.selectionLimit`, and `improvementDiscovery.userNarrative` in the report narrative. `candidates[]` carries the same 11 logical fields described above; `lensCoverage[]` records either candidate IDs or an evidence-backed no-candidate rationale for every resolved lens. `schemas/final-report-v3.0.schema.json`, `schemas/report-narrative-v3.0.schema.json`, and `validators/validate_improvement_report.py` enforce this contract. Phase 7 assembles the final record and derives the reading copy and human HTML.
|
|
80
80
|
- Clarification request policy (phase-specific addenda — shared policy is in `_common-contract.md`):
|
|
81
81
|
- if scan-scope or priority-lenses cannot be made concrete during Phase 1.5, end the run with Verdict Token `blocked`, populate `## 1. Clarification Items` with `Blocks=next-phase` rows, and do not run worker dispatch
|
|
82
82
|
{{INCLUDE:_clarification-recommendation.md}}
|
|
@@ -158,7 +158,8 @@ For a `host-text` mapping, render each numbered item as its option label followe
|
|
|
158
158
|
- CLI-wrapper assignments never enter the Agent layer. `okstra worker-dispatch` validates the metadata and passes `modelExecutionValue` to the registered provider script.
|
|
159
159
|
- Missing or unsupported family-token mapping is a pre-dispatch contract failure. Never inherit the lead model, choose a nearby alias, or switch provider silently.
|
|
160
160
|
- Every analysis dispatch sets `name: "<workerId>-worker"`; convergence retries append `-reverify-r<N>`, implementation uses the functional `-executor` / `-verifier` suffix, and report writing uses `report-writer`. These values are retained as `agentName` in session JSONL for usage attribution.
|
|
161
|
-
-
|
|
161
|
+
- A CLI worker's role reaches the entrypoint from the invocation's duty, NOT from the prompt text. `WorkerJob.wrapper_role` resolves `role_for_duty(dutyId)` and passes the canonical role in the role positional; the entrypoint checks it against the same derivation in the invocation metadata and refuses a mismatch. Do not add, edit, or rely on a `**Pane role:**` line to change what a dispatch runs as — the prompt body carries that header for the initial analysis audiences only, and it selects nothing.
|
|
162
|
+
- The role decides the dispatch's idle budget, and `worker-dispatch` leaves the budget positional EMPTY so it can: an empty slot makes the entrypoint read the role's own budget (`domain/worker_role.role_spec` — `implementer` and `verifier` run silent build+test suites and get 1500s, every other role 600s). Passing an explicit `--idle-timeout-seconds` overrides that for the whole dispatch, so pass one only for a run-specific reason; filling the slot with a blanket default is what made the role budgets unreachable and reaped healthy workers mid-suite.
|
|
162
163
|
- The host may supply transport metadata for native calls, but acceptance records only the verified invocation specification link. Record `enforcementMode=host-native-spec-link-gate`, `promptPath`, and `metadataPath`; do not claim the host-delivered bytes were observed.
|
|
163
164
|
- A retry keeps the same Agent `name`. When logging a twice-failed CLI-wrapper attempt, reference both attempts' `bash_ids` and prompt-history paths.
|
|
164
165
|
- An internally detected contract violation without a specific worker uses `--agent "claude-lead"` in the error-log event.
|
|
@@ -91,7 +91,7 @@ Render every numbered item as its option label followed by its description verba
|
|
|
91
91
|
- Worker completion is valid only from `workerDispatches[]`, terminal status sidecars, and required Result Paths. Pane creation alone is not completion.
|
|
92
92
|
- Reverify uses a fresh jobs file at `runs/<task-type>/state/reverify-jobs-r<N>-<task-type>-<seq>.json`, sets `dispatchKind: "reverify-r<N>"`, and dispatches with `okstra team dispatch --project-root <root> --run-manifest <path> --dispatch-kind reverify-r<N> --jobs-file <jobs-file>`.
|
|
93
93
|
- Report-writer uses a fresh one-job jobs file with `dispatchKind: "report-writer"` and the same schema, then dispatches through `okstra team dispatch --project-root <root> --run-manifest <path> --jobs-file <jobs-file>`.
|
|
94
|
-
- Every reverify or report-writer jobs file carries `workerId`, `provider`, `role`, `modelExecutionValue`, `promptPath`, `promptMetadataPath`, `invocationId`, `assignmentRef`, `audience`, the five prompt digests, `resultPath`, `workerResultPath`, and `completionPaths`. `worker-dispatch` verifies these fields before launching the provider process. For reverify, set `role` to `worker-reverify-r<N>`. The report-writer completion paths include
|
|
94
|
+
- Every reverify or report-writer jobs file carries `workerId`, `provider`, `role`, `modelExecutionValue`, `promptPath`, `promptMetadataPath`, `invocationId`, `assignmentRef`, `audience`, the five prompt digests, `resultPath`, `workerResultPath`, and `completionPaths`. `worker-dispatch` verifies these fields before launching the provider process. For reverify, set `role` to `worker-reverify-r<N>`. The report-writer completion paths include its narrative Markdown, worker-result pointer, and audit sidecar; Phase 7 later assembles `data.json`.
|
|
95
95
|
- After either dispatch, run `okstra team await --project-root <root> --run-manifest <path>` before evaluating terminal status or completion paths.
|
|
96
96
|
|
|
97
97
|
## Completion, cleanup, and resume
|
|
@@ -116,6 +116,14 @@ def _activity_row(event: LeadEvent) -> dict[str, Any]:
|
|
|
116
116
|
from okstra_ctl.execution_identity import stored_identity
|
|
117
117
|
|
|
118
118
|
row = {key: details[key] for key in ACTIVITY_FIELDS}
|
|
119
|
+
if "clarificationRefs" in details:
|
|
120
|
+
refs = details["clarificationRefs"]
|
|
121
|
+
if not isinstance(refs, list) or any(
|
|
122
|
+
not isinstance(ref, str) or re.fullmatch(r"C-\d{3,}", ref) is None
|
|
123
|
+
for ref in refs
|
|
124
|
+
):
|
|
125
|
+
raise ActivityProjectionError("invalid clarificationRefs")
|
|
126
|
+
row["clarificationRefs"] = list(refs)
|
|
119
127
|
row.update(stored_identity(details))
|
|
120
128
|
return row
|
|
121
129
|
|
|
@@ -179,12 +187,11 @@ def record_activity(
|
|
|
179
187
|
return append_activity_event(events_path, event)
|
|
180
188
|
|
|
181
189
|
|
|
182
|
-
def
|
|
190
|
+
def agent_activity_rows(
|
|
183
191
|
project_root: Path,
|
|
184
192
|
run_manifest_path: Path,
|
|
185
|
-
data_path: Path,
|
|
186
193
|
) -> tuple[dict[str, Any], ...]:
|
|
187
|
-
"""
|
|
194
|
+
"""현재 실행의 정본 활동을 파일 변경 없이 반환한다."""
|
|
188
195
|
manifest = _read_json_object(run_manifest_path)
|
|
189
196
|
if manifest.get("activityContractVersion") != 1:
|
|
190
197
|
return ()
|
|
@@ -204,6 +211,19 @@ def project_agent_activity(
|
|
|
204
211
|
)
|
|
205
212
|
rows = tuple(_activity_row(event) for event in events)
|
|
206
213
|
_validate_activity_order(rows)
|
|
214
|
+
return rows
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def project_agent_activity(
|
|
218
|
+
project_root: Path,
|
|
219
|
+
run_manifest_path: Path,
|
|
220
|
+
data_path: Path,
|
|
221
|
+
) -> tuple[dict[str, Any], ...]:
|
|
222
|
+
"""2.0 호환 경로에서 ``agentActivity``만 교체한다."""
|
|
223
|
+
rows = agent_activity_rows(project_root, run_manifest_path)
|
|
224
|
+
manifest = _read_json_object(run_manifest_path)
|
|
225
|
+
if manifest.get("activityContractVersion") != 1:
|
|
226
|
+
return ()
|
|
207
227
|
data = _read_json_object(data_path)
|
|
208
228
|
data["agentActivity"] = list(rows)
|
|
209
229
|
_atomic_write_json(data_path, data)
|
|
@@ -48,7 +48,10 @@ from .wrapper_status import (
|
|
|
48
48
|
prompt_derived_paths,
|
|
49
49
|
status_path_for_prompt,
|
|
50
50
|
)
|
|
51
|
-
from .worker_prompt_policy import
|
|
51
|
+
from .worker_prompt_policy import (
|
|
52
|
+
CRITIC_DUTY_BY_ASSIGNMENT_SEGMENT,
|
|
53
|
+
resolve_prompt_plan_for_manifest,
|
|
54
|
+
)
|
|
52
55
|
from .dispatch_state import (
|
|
53
56
|
BACKEND_CLI_WRAPPER,
|
|
54
57
|
BACKEND_CMUX_PANE,
|
|
@@ -314,7 +317,7 @@ def _materialize_run(
|
|
|
314
317
|
)
|
|
315
318
|
if args.audience == "report-writer" and not args.audit_source:
|
|
316
319
|
# The report writer is the one audience whose result path is not its own
|
|
317
|
-
# worker result: it writes the report
|
|
320
|
+
# worker result: it writes the report narrative, while the audit
|
|
318
321
|
# sidecar is derived from its `.md`. With both collapsed into one value
|
|
319
322
|
# the prompt loses its `**Worker Result Path:**` anchor and the writer
|
|
320
323
|
# puts the report where the audit file belongs — silently, because every
|
|
@@ -630,10 +633,7 @@ def _validate_run_identity(
|
|
|
630
633
|
raise AgentPromptCliError("translator assignment requires worker ID 'translator'")
|
|
631
634
|
elif assignment_ref.startswith("critic/"):
|
|
632
635
|
scope = assignment_ref.split("/", 1)[1]
|
|
633
|
-
expected =
|
|
634
|
-
scope,
|
|
635
|
-
"",
|
|
636
|
-
)
|
|
636
|
+
expected = CRITIC_DUTY_BY_ASSIGNMENT_SEGMENT.get(scope, "")
|
|
637
637
|
if not expected or dispatch_kind != "critic":
|
|
638
638
|
raise AgentPromptCliError("critic assignment identity is invalid")
|
|
639
639
|
convergence = manifest.get("convergence")
|
|
@@ -101,6 +101,7 @@ def build_analysis_packet(
|
|
|
101
101
|
instruction_set_relative_path: str,
|
|
102
102
|
fix_history_text: str = "",
|
|
103
103
|
stage_ledger_json: str = "",
|
|
104
|
+
stage_ledger_notice: str = "",
|
|
104
105
|
) -> str:
|
|
105
106
|
"""Return the primary compact input for Claude/Codex/Antigravity analysers."""
|
|
106
107
|
brief_text = task_brief_path.read_text(encoding="utf-8")
|
|
@@ -124,6 +125,7 @@ def build_analysis_packet(
|
|
|
124
125
|
parts.extend(_reference_block(reference_text))
|
|
125
126
|
parts.extend(_fix_history_block(fix_history_text))
|
|
126
127
|
parts.extend(_stage_ledger_block(stage_ledger_json))
|
|
128
|
+
parts.extend(_stage_ledger_unavailable_block(stage_ledger_notice))
|
|
127
129
|
parts.extend(_clarification_block(clarification_text))
|
|
128
130
|
parts.extend(_directive_block(directive))
|
|
129
131
|
return "\n".join(part.rstrip() for part in parts).rstrip() + "\n"
|
|
@@ -240,8 +242,20 @@ def _stage_ledger_block(stage_ledger_json: str) -> list[str]:
|
|
|
240
242
|
"",
|
|
241
243
|
"Facts about this task's stages as they stand on disk — not a plan.",
|
|
242
244
|
"A stage whose `status` is `done` is already implemented and will not",
|
|
243
|
-
"be executed again, so do not rewrite its plan body.
|
|
244
|
-
"
|
|
245
|
+
"be executed again, so do not rewrite its plan body.",
|
|
246
|
+
"",
|
|
247
|
+
"`stages` lists every stage the latest plan declares, so every number",
|
|
248
|
+
"in it is taken. Never reuse or renumber one: a new stage takes the",
|
|
249
|
+
"next number after the highest listed here, and reworking a completed",
|
|
250
|
+
"stage means cancelling it and adding a new number, never editing it",
|
|
251
|
+
"in place. `sourcePlan` is the plan the completed stages were built",
|
|
252
|
+
"against; `latestPlan` is the plan this list came from. When the two",
|
|
253
|
+
"differ, the completed work followed the former and the numbering",
|
|
254
|
+
"authority is the latter.",
|
|
255
|
+
"",
|
|
256
|
+
"A `planDivergence` entry means the two plans disagree about a",
|
|
257
|
+
"completed stage. Treat it as a blocker for any new stage number and",
|
|
258
|
+
"report it; do not resolve it by choosing one of the two yourself.",
|
|
245
259
|
"",
|
|
246
260
|
"```json",
|
|
247
261
|
stage_ledger_json.strip(),
|
|
@@ -250,6 +264,33 @@ def _stage_ledger_block(stage_ledger_json: str) -> list[str]:
|
|
|
250
264
|
]
|
|
251
265
|
|
|
252
266
|
|
|
267
|
+
def _stage_ledger_unavailable_block(notice: str) -> list[str]:
|
|
268
|
+
"""원장을 못 읽었다는 사실을 packet 에 남긴다.
|
|
269
|
+
|
|
270
|
+
블록을 생략하면 저작 쪽은 "이전 계획이 없다" 로 읽는다 — 프로파일이 그렇게
|
|
271
|
+
지시한다. 계획이 있는데 못 읽은 경우에 그 침묵은 거짓이고, 이미 점유된
|
|
272
|
+
번호를 새 stage 에 다시 내주는 경로가 된다.
|
|
273
|
+
"""
|
|
274
|
+
if not notice.strip():
|
|
275
|
+
return []
|
|
276
|
+
return [
|
|
277
|
+
"",
|
|
278
|
+
"## Stage Ledger",
|
|
279
|
+
"",
|
|
280
|
+
"This task's stage facts could NOT be read, so no ledger is included.",
|
|
281
|
+
"Do not read this as 'the task has no prior plan' — a plan exists and",
|
|
282
|
+
"this run could not parse it.",
|
|
283
|
+
"",
|
|
284
|
+
f"- Reason: {notice.strip()}",
|
|
285
|
+
"",
|
|
286
|
+
"Do not assign a number to any new stage in this state. A number taken",
|
|
287
|
+
"by a plan this run could not read would collide with the completed",
|
|
288
|
+
"work under it, and that collision stays silent until integration.",
|
|
289
|
+
"Report this as a blocker instead.",
|
|
290
|
+
"",
|
|
291
|
+
]
|
|
292
|
+
|
|
293
|
+
|
|
253
294
|
def _clarification_block(clarification_text: str) -> list[str]:
|
|
254
295
|
if not clarification_text.strip():
|
|
255
296
|
return []
|