okstra 0.209.5 → 0.210.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (23) hide show
  1. package/package.json +1 -1
  2. package/runtime/BUILD.json +2 -2
  3. package/runtime/prompts/wizard/prompts.ko.json +4 -3
  4. package/runtime/python/okstra_ctl/analysis_packet.py +7 -3
  5. package/runtime/python/okstra_ctl/incremental_scope.py +11 -2
  6. package/runtime/python/okstra_ctl/phases/final_verification/target.py +5 -0
  7. package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-executor.md +1 -1
  8. package/runtime/python/okstra_ctl/phases/implementation/profile.md +1 -1
  9. package/runtime/python/okstra_ctl/phases/implementation/wizard.py +1 -0
  10. package/runtime/python/okstra_ctl/phases/implementation_planning/authoring.py +2 -1
  11. package/runtime/python/okstra_ctl/phases/implementation_planning/profile.md +3 -3
  12. package/runtime/python/okstra_ctl/phases/implementation_planning/report_assets/implementation-planning-input.template.md +1 -1
  13. package/runtime/python/okstra_ctl/phases/release_handoff/operations.py +6 -0
  14. package/runtime/python/okstra_ctl/plan_items.py +24 -3
  15. package/runtime/python/okstra_ctl/stage_close.py +13 -2
  16. package/runtime/python/okstra_ctl/stage_map.py +25 -0
  17. package/runtime/python/okstra_ctl/stage_map_cli.py +3 -2
  18. package/runtime/python/okstra_ctl/stage_targets.py +17 -3
  19. package/runtime/python/okstra_ctl/wizard/sources.py +2 -1
  20. package/runtime/python/okstra_ctl/wizard/steps_plan.py +4 -2
  21. package/runtime/python/okstra_ctl/write_policy.py +10 -3
  22. package/runtime/schemas/final-report-v3.0.schema.json +7 -0
  23. package/runtime/validators/validate-implementation-plan-stages.py +46 -5
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "okstra",
3
- "version": "0.209.5",
3
+ "version": "0.210.0",
4
4
  "description": "Host-aware multi-provider cross-verification orchestrator runtime and agent skills.",
5
5
  "license": "MIT",
6
6
  "author": "devonshin",
@@ -1,5 +1,5 @@
1
1
  {
2
- "package": "0.209.5",
3
- "builtAt": "2026-10-05T07:34:52.243Z",
2
+ "package": "0.210.0",
3
+ "builtAt": "2026-10-05T13:33:31.081Z",
4
4
  "repoRoot": "/home/runner/work/okstra/okstra"
5
5
  }
@@ -354,16 +354,17 @@
354
354
  "mark_done": "[완료]",
355
355
  "mark_active": "[진행중]",
356
356
  "mark_ready": "[준비됨]",
357
- "mark_blocked": "[대기]"
357
+ "mark_blocked": "[대기]",
358
+ "mark_cancelled": "[취소됨]"
358
359
  },
359
360
  "errors": {
360
361
  "none_selected": "stage 를 하나 이상 선택하세요.",
361
362
  "whole_task_impl": "전체 task 검증은 final-verification 에서만 선택할 수 있습니다.",
362
363
  "all_exclusive": "'전체' 는 단독으로만 선택할 수 있습니다.",
363
- "nothing_selectable": "선택 가능한 stage 가 없습니다 (모두 완료/진행중).",
364
+ "nothing_selectable": "선택 가능한 stage 가 없습니다 (모두 완료/진행중/취소됨).",
364
365
  "bad_number": "stage 번호가 올바르지 않습니다: {answer}",
365
366
  "unknown_stage": "Stage Map 에 없는 stage 입니다: {bad}",
366
- "occupied": "이미 완료/진행중인 stage 는 선택할 수 없습니다: {bad}"
367
+ "occupied": "이미 완료/진행중이거나 취소된 stage 는 선택할 수 없습니다: {bad}"
367
368
  },
368
369
  "echo_variants": {
369
370
  "plain": "stages: {stages}",
@@ -535,9 +535,13 @@ def _stage_ledger_block(stage_ledger_json: str) -> list[str]:
535
535
  "",
536
536
  "`stages` lists every stage the latest plan declares, so every number",
537
537
  "in it is taken. Never reuse or renumber one: a new stage takes the",
538
- "next number after the highest listed here, and reworking a completed",
539
- "stage means cancelling it and adding a new number, never editing it",
540
- "in place. The new plan declares every stage listed here as well as the",
538
+ "next number after the highest listed here. Changing a `done` stage's",
539
+ "work means keeping it and adding a correction stage with a new number,",
540
+ "never editing it in place. Only a `ready` or `blocked` stage may be",
541
+ "cancelled: keep its row and body where they are and set its",
542
+ "`stageMap[]` row's `status` to `cancelled`. A `done` or `active` stage",
543
+ "is never cancelled, and no remaining stage may depend on a cancelled",
544
+ "one. The new plan declares every stage listed here as well as the",
541
545
  "ones it adds — its rows run 1..N with no gap — because a plan whose",
542
546
  "rows start above 1 is refused by every consumer that parses a Stage",
543
547
  "Map. `sourcePlan` is the plan the completed stages were built",
@@ -102,8 +102,17 @@ def _parse_depends_on(cell: str) -> list[int]:
102
102
 
103
103
 
104
104
  def parse_stage_graph(data: dict) -> list[tuple[int, list[int]]]:
105
+ """재검증·이월 대상이 될 수 있는 stage 그래프 — 취소 행은 뺀다.
106
+
107
+ 취소 stage 는 다시 계획되지도 이월되지도 않는다. 세면 컷오프 분모가 부풀어
108
+ 좁혀야 할 재실행이 증분으로 남거나, 지목할 수 없는 번호가 선택지에 오른다.
109
+ """
105
110
  stage_map = data.get("implementationPlanning", {}).get("stageMap", [])
106
- return [(int(row["stage"]), _parse_depends_on(row.get("dependsOn", ""))) for row in stage_map]
111
+ return [
112
+ (int(row["stage"]), _parse_depends_on(row.get("dependsOn", "")))
113
+ for row in stage_map
114
+ if row.get("status") != "cancelled"
115
+ ]
107
116
 
108
117
 
109
118
  def design_prep_impacted_stages(data: dict, item_ids: set[str]) -> set[int]:
@@ -363,7 +372,7 @@ def decide_scope(
363
372
  )
364
373
  all_stages = {num for num, _ in stages}
365
374
  closure = downstream_stage_closure(stages, set(impacted_stages))
366
- if len(closure) * 2 > len(all_stages):
375
+ if len(closure) > cutoff_ratio * len(all_stages):
367
376
  return IncrementalDecision(
368
377
  "full", [], [],
369
378
  f"the answers reach {len(closure)} of the plan's {len(all_stages)} stages — "
@@ -24,6 +24,7 @@ from okstra_ctl.locks import worktree_provision_mutex
24
24
  from okstra_ctl.plan_run_root import plan_run_root_from_approved_plan
25
25
  from okstra_ctl.prepare_error import PrepareError
26
26
  from okstra_ctl.stage_integrate import IntegrateResult
27
+ from okstra_ctl.stage_map import active_stage_records
27
28
  from okstra_ctl.stage_reconcile import auto_reconcile_best_effort
28
29
  from okstra_ctl.stage_targets import (
29
30
  FinalVerificationTarget,
@@ -274,6 +275,10 @@ def acquire_final_verification_target(
274
275
  slugify(request.task_group),
275
276
  slugify(request.task_id),
276
277
  ):
278
+ # 취소된 stage 는 완료를 요구받지 않고 통합·tip 선택에서도 빠진다.
279
+ request = replace(
280
+ request, stage_map=tuple(active_stage_records(request.stage_map)),
281
+ )
277
282
  registry_coordinates = _final_verification_registry_coordinates(request)
278
283
  done_rows = _read_final_verification_done_rows(
279
284
  request,
@@ -89,7 +89,7 @@ template's check; that template is gone.
89
89
  ```
90
90
 
91
91
  The file MUST NOT exist before the run starts (overwrite is refused — see `--force-stage` non-goal). **Enforced:** `validators/validate-run.py` `_validate_stage_carry_sidecar_exists` fails a run that declares `stageSidecarEvidence` without the file on disk. Transcribing the JSON into the report is not the same as writing it: `consumers` treats the carry file as the source of truth for marking the stage `done`, so a missing file leaves the stage permanently incomplete and blocks every dependent stage with a `PrepareError` — while this run reports success.
92
- - **Verifier gates are not yours to run (BLOCKING).** The self-mock detector (`validators/detect_self_mock.py`) belongs to the implementation verifier and is never delegated to you (`_implementation-verifier.md` §"Self-mock detection"). The coding-conventions preflight names it as the enforcement behind the no-self-mocking principle — that names who will check your diff, not a command for you to run. You MUST NOT invoke it and MUST NOT write `<task_root>/qa/self-mock-*.json`; running it early does not pre-satisfy the gate, because the verifier runs it again under its own duty. **Enforced:** that sidecar is not among the paths your attempt's `writePolicy.artifactPolicy.allowedPaths` carries, so the write audit closes the attempt as `error` with `artifact-root change exceeds batch policy union` (`scripts/okstra_ctl/execution_mutation_audit.py`) — a stage whose every gate passed still lands as a failed run. The same holds for the conformance results `<task_root>/qa/result-*.json`: you may run a conformance script to check your work — whatever it writes (a baseline capture, build logs) goes under `<task_root>/qa/output/`, the one qa directory besides `qa/scripts/` your attempt may write — but the verifier records every result, including an earlier stage's result that your rewrite of its script made stale. When the approved plan tells you to write one, leave it and name it in your result as a plan step the verifier owns.
92
+ - **Verifier gates are not yours to run (BLOCKING).** The self-mock detector (`validators/detect_self_mock.py`) belongs to the implementation verifier and is never delegated to you (`_implementation-verifier.md` §"Self-mock detection"). The coding-conventions preflight names it as the enforcement behind the no-self-mocking principle — that names who will check your diff, not a command for you to run. You MUST NOT invoke it and MUST NOT write `<task_root>/qa/self-mock-*.json`; running it early does not pre-satisfy the gate, because the verifier runs it again under its own duty. **Enforced:** that sidecar is not among the paths your attempt's `writePolicy.artifactPolicy.allowedPaths` carries, so the write audit closes the attempt as `error` with `artifact-root change exceeds batch policy union` (`scripts/okstra_ctl/execution_mutation_audit.py`) — a stage whose every gate passed still lands as a failed run. The same holds for the conformance results `<task_root>/qa/result-*.json`: you may run a conformance script to check your work — a baseline capture or `--compare` target goes under `<task_root>/qa/baseline/` and anything else it writes (build logs, measurements) under `<task_root>/qa/output/`, the only qa directories besides `qa/scripts/` your attempt may write — but the verifier records every result, including an earlier stage's result that your rewrite of its script made stale. When the approved plan tells you to write one, leave it and name it in your result as a plan step the verifier owns.
93
93
  - **An external Tier 3 non-PASS does NOT withhold the carry evidence.** A Tier 3 entry whose `requires` include `http`, `external`, or `db` is advisory. Its FAIL, MISSING, no result, startup failure, or credential / network / service absence gets recorded honestly — exact command, exit code, output tail, marked `ADVISORY` in `Validation evidence` — and you emit the carry evidence anyway. Only Tier 1 and Tier 2 failures withhold it. Withholding on an external result is what actually blocks the stage: the carry file is the only thing that can mark a stage `done`, the verifier re-runs that same command from the host (where a call your sandbox could not complete often passes), and a stage the verifier then PASSes can never be closed because its evidence was never written.
94
94
  - **Reverse link (BLOCKING).** The runtime already appended a `status:"started"` row for this stage before the run began. The terminal row belongs to the lead's post-stage persistence and is verdict-gated — `status:"done"` with `carry_path` on a non-`FAIL` verdict, `status:"failed"` on `FAIL` (`_implementation-deliverable.md` §"Lead post-stage persistence").
95
95
  - **No PR / push in this phase.** This run produces local commits, carry sidecar evidence, verifier results, and the implementation final report only. Push and PR creation belong exclusively to the later `release-handoff` phase after `final-verification` returns `accepted`.
@@ -2,7 +2,7 @@
2
2
 
3
3
  - Purpose: realise the approved `implementation-planning` deliverable as actual source changes, with cross-model verification, while keeping the run reversible
4
4
  - **Run-level fixed cost:** the verifier set, Phase 5.5 convergence, and the Phase 6 report-writer run exactly once per implementation run, over this run's single stage diff — never once per step.
5
- - **Fix run (profile carries a "Fix-Run Carry" block):** the executor's scope is the carried blocking findings plus the previous routing recommendation — it MUST NOT re-execute plan steps the previous run completed. Verifiers apply the "Fix-run incremental scope" section of `_implementation-verifier.md`; the report writer applies "Fix-run incremental authoring" in `report-writer.md`. The full validation-command re-run is NOT reduced.
5
+ - **Fix run (profile carries a "Fix-Run Carry" block):** the executor's scope is the carried blocking findings plus the previous routing recommendation — it MUST NOT re-execute plan steps the previous run completed. Verifiers apply the "Fix-run incremental scope" section of `_implementation-verifier.md`. The full validation-command re-run is NOT reduced.
6
6
  - **Executor binding (resolved at run-prep time, fixed for this run):**
7
7
  - Executor display name: `{{EXECUTOR_DISPLAY_NAME}}`
8
8
  - Executor worker ID: `{{EXECUTOR_WORKER_ID}}`
@@ -9,6 +9,7 @@ _STAGE_MARKER_KEYS = {
9
9
  "active": "mark_active",
10
10
  "ready": "mark_ready",
11
11
  "blocked": "mark_blocked",
12
+ "cancelled": "mark_cancelled",
12
13
  }
13
14
 
14
15
 
@@ -1284,7 +1284,8 @@ def _stage_ledger_snapshot(run_manifest: Path | None) -> dict[str, str] | None:
1284
1284
  """`{스테이지 번호: 상태}`. 원장을 못 읽으면 ``None``.
1285
1285
 
1286
1286
  게이트가 범위를 좁히려면 어느 스테이지가 지금 시작 가능한지 알아야 한다. 그
1287
- 사실은 Stage 원장에 있고, 상태 어휘(`done` / `active` / `ready` / `blocked`)는
1287
+ 사실은 Stage 원장에 있고, 상태 어휘(`done` / `cancelled` / `active` / `ready` /
1288
+ `blocked`)는
1288
1289
  `stage_targets.StageLifecycle.status` 의 것을 그대로 쓴다 — 원장이 자기 어휘를
1289
1290
  따로 가지면 같은 stage 가 소비처마다 다르게 읽힌다.
1290
1291
 
@@ -42,9 +42,9 @@ Plan for the actual worktree layout before approval. Shared documentation direct
42
42
  - **directive-first ambiguity resolution** (the same rule, pointed at the user instead of the code): any ambiguity the run's directive, the brief, the carried-in `user-responses/` sidecars, or the user's in-session instruction already answers MUST be resolved that way and recorded with the quoted instruction. Writing a clarification row for something the user already decided is the same defect as writing one for something the code already answers — and it costs more, because the row withholds approval until a whole separate answer cycle closes it. When an instruction points at a document, treat every item in that document as decided, including the ones the document itself flagged as needing a decision (shared rule: `_common-contract.md` "User instruction outranks the material it points at").
43
43
  - flag any requirement that is ambiguous, contradictory, or missing success criteria — register each one as a row in the report's `## 1. Clarification Items` table with `Blocks=approval` instead of guessing
44
44
  - read `<PROJECT_ROOT>/.okstra/glossary.md` and `<PROJECT_ROOT>/.okstra/decisions/` titles if present. Absent okstra memory files are the normal state — do not error. Treat the brief's `terminology:*` resolutions from `requirements-discovery` (if any) as authoritative; if missing, resolve any remaining fuzzy term as a `Blocks=approval` clarification row.
45
- - **Stage Ledger (read before drafting the Stage Map):** when this task already has a plan on disk, the analysis packet carries a `## Stage Ledger` JSON block listing every stage with its `status` (`done` / `active` / `ready` / `blocked`), `dependsOn`, and done commit. It states what exists, not what to plan. Two rules follow from it:
45
+ - **Stage Ledger (read before drafting the Stage Map):** when this task already has a plan on disk, the analysis packet carries a `## Stage Ledger` JSON block listing every stage with its `status` (`done` / `active` / `cancelled` / `ready` / `blocked`), `dependsOn`, and done commit. It states what exists, not what to plan. Two rules follow from it:
46
46
  - A stage whose `status` is `done` is already implemented and will not be executed again. Declare it in this report's `stages[]` and `stageMap[]` with its plan body copied forward as written; do not rewrite its steps, and do not fold its work into a new stage. Its plan items are `observed`, not `in-scope` (`okstra_ctl.plan_items.stage_scope_bucket`), so re-declaring it adds no verification load. A carried body is also not re-judged: the planning conformance gate skips a stage the consumer ledger records as `done`, and report assembly issues no design-prep request for one (`okstra_ctl.design_prep.materialize_design_prep_requests`). A rule that widened since that stage shipped cannot be satisfied by a body you are forbidden to rewrite.
47
- - Every stage number in the ledger is taken. This report declares every one of them — rows run `1..N` with no gap — and a new stage takes the next number after the highest one listed; numbers are never reused or reordered. A report that declares only its new stages satisfies neither rule and is refused twice over. **Enforced:** `okstra_ctl.stage_map._validate_stage_numbers` rejects the gap in every published plan it parses, so `okstra prepare` refuses the implementation run (`okstra_ctl.run._parse_stage_map_into_ctx`) and the Stage Ledger of every later run reads as `unreadable`; `validators/validate-implementation-plan-stages.py` check **S2** reports the same defect at authoring time. Cancelling a stage you no longer want is not yet expressible — that lands with the plan-amendment feature — so an unstarted stage you drop still keeps its number and body here.
47
+ - Every stage number in the ledger is taken. This report declares every one of them — rows run `1..N` with no gap — and a new stage takes the next number after the highest one listed; numbers are never reused or reordered. A report that declares only its new stages satisfies neither rule and is refused twice over. **Enforced:** `okstra_ctl.stage_map._validate_stage_numbers` rejects the gap in every published plan it parses, so `okstra prepare` refuses the implementation run (`okstra_ctl.run._parse_stage_map_into_ctx`) and the Stage Ledger of every later run reads as `unreadable`; `validators/validate-implementation-plan-stages.py` check **S2** reports the same defect at authoring time. To drop a stage that has not started — `ready` or `blocked` in the ledger — keep its row and body in place in both `stageMap[]` and `stages[]` and set that `stageMap[]` row's `status` to `cancelled`; its number stays taken, and stage selection, whole-task final-verification, integration and release-handoff skip it. A `done` or `active` stage is never cancelled: to change finished work, keep that stage and add a correction stage numbered after the highest one. No remaining stage may depend on a cancelled one — point it at the stage that replaces it. **Enforced:** `validators/validate-implementation-plan-stages.py` check **S14** rejects a cancelled stage the ledger records `done` or `active` and a dependency on a cancelled stage.
48
48
  - The ledger answers two questions from two sources, and the block names both. `sourcePlan` is the plan the completed stages were actually built against; `latestPlan` is the plan the `stages` list came from and is therefore the numbering authority. When they differ, the completed work followed the former and the highest taken number comes from the latter.
49
49
  - A `planDivergence` entry means one of two things: the two plans disagree about a stage that is already `done` — the same number naming different work, or a completed stage the latest plan no longer declares — or the plan the completed stages were built against could not be read at all, so that comparison never ran. Both block the same way: do not pick one of the two plans yourself; register it as a `Blocks=approval` clarification row and assign no new stage number until it is resolved. Completed stages built against an *earlier* plan are not a divergence — that is the normal shape of an amended plan and the ledger folds it silently.
50
50
  - The block is absent ONLY on a task's first planning run. Its absence then means there is no prior plan, not that no stage is done. When the ledger could not be read, the packet names the source and reason. Continue planning to repair unambiguous dependency notation in the new report, retaining every stage number and title and every completed stage body. Record the original and corrected values with their source. Do not overwrite the prior report or select an older plan. Assign no new stage number until the corrected Stage Map validates; unresolved dependency meaning remains a blocker. The preparation behavior is covered by `tests/run/test_stage_ledger_prepare.py`; preservation of author intent remains a review guideline.
@@ -160,7 +160,7 @@ Plan for the actual worktree layout before approval. Shared documentation direct
160
160
  unavailable environment is a user-owned follow-up, never a plan approval or
161
161
  later run blocker. `requires=[]` and `requires=[io]` remain blocking.
162
162
  Remote IO should also declare `external`.
163
- Layout split (not this phase's writes): executable scripts (conformance + any real-IO test) live under `<task_root>/qa/scripts/`; data sidecars (`conformance-manifest.json`, `result-*.json`) stay at the `qa/` root. A step that runs a script writing its own files (a baseline capture, a `--compare` target, build logs) points those paths at `<task_root>/qa/output/`: it is the only other qa directory the implementer's write audit accepts, and any other qa path discards the attempt (`scripts/okstra_ctl/write_policy.py` `_role_qa_artifact_paths`; observed 2026-09-26, dev-11054 stage 1, `qa/baseline/`). The implementer writes the scripts and the manifest entry. Only the implementation verifier writes `result-*.json`, including the result of an earlier stage's entry whose script this stage rewrites, so a step never assigns a result file to the implementer and never lists one in `plannedPaths`. This declaration is enforced at four layers: `validators/validate-implementation-plan-stages.py` check **S11** forces every stage to carry one of the two lines; at the planning boundary `scripts/okstra_ctl/phases/implementation_planning/plan_body.py` `_validate_planning_conformance_declared` accepts a well-formed `Conformance tests:` line even when the script file and manifest entry are absent (malformed `requires` still fails); the matching `implementation` stage run that inherited `Conformance tests:` fails closed when the script file is missing (`_validate_conformance`); and the manifest JSON structure — including each entry's `script` living under `qa/scripts/` and a `runCommand` that does not change cwd — is enforced by `validate_conformance_manifest` when the implementer writes the entry.
163
+ Layout split (not this phase's writes): executable scripts (conformance + any real-IO test) live under `<task_root>/qa/scripts/`; data sidecars (`conformance-manifest.json`, `result-*.json`) stay at the `qa/` root. A step that captures a baseline or a `--compare` target points it at `<task_root>/qa/baseline/`; anything else a script writes on every run (build logs, measurements, captured responses) goes under `<task_root>/qa/output/`. These are the only other qa directories the implementer's write audit accepts, and any other qa path discards the attempt (`scripts/okstra_ctl/write_policy.py` `_role_qa_artifact_paths`). The split matters at final verification: the verifier re-runs every Tier 3 script and may rewrite `qa/output/`, but `qa/baseline/` must stay byte-identical (`_inherited_qa_evidence`), so a baseline written under `qa/output/` is not protected. The implementer writes the scripts and the manifest entry. Only the implementation verifier writes `result-*.json`, including the result of an earlier stage's entry whose script this stage rewrites, so a step never assigns a result file to the implementer and never lists one in `plannedPaths`. This declaration is enforced at four layers: `validators/validate-implementation-plan-stages.py` check **S11** forces every stage to carry one of the two lines; at the planning boundary `scripts/okstra_ctl/phases/implementation_planning/plan_body.py` `_validate_planning_conformance_declared` accepts a well-formed `Conformance tests:` line even when the script file and manifest entry are absent (malformed `requires` still fails); the matching `implementation` stage run that inherited `Conformance tests:` fails closed when the script file is missing (`_validate_conformance`); and the manifest JSON structure — including each entry's `script` living under `qa/scripts/` and a `runCommand` that does not change cwd — is enforced by `validate_conformance_manifest` when the implementer writes the entry.
164
164
  - `### Stage Exit Contract` — predicted added/modified files, newly exposed identifiers/types/endpoints, downstream-usable resources.
165
165
  - `### Stage Validation` — pre / mid / post exact commands or observable outcomes for this stage only.
166
166
  - **Run-executable only (BLOCKING).** A `validationChecklist` row that carries `stageRefs` gates that stage's carry, so it MUST be executable by the implementation run itself — no deployment, no manual walk-through, no observation "by hand", no credentials the run does not hold. The implementation phase forbids deploys and holds no deploy credentials, so such a row can never pass and blocks a stage with zero code defects (observed: dev-10341 VC-013/VC-014). Manual or deployed-environment verification belongs in the brief's `External Gates` and final-verification's user-owned external QA — record it there without `stageRefs`. **Enforced:** `okstra_ctl.implementation_direction.validate_selected_direction_plan` (`stage_validation_executability_errors`) rejects stage-gating rows whose observation text requires a manual or deployed-environment step.
@@ -127,7 +127,7 @@ Enforced: `scripts/okstra_ctl/run.py` `_validate_approved_plan` reads that front
127
127
 
128
128
  ## Stage Output Shape (reference)
129
129
 
130
- **Enforced:** `validators/validate-implementation-plan-stages.py` fails a plan missing `## 5.5 Stage Map` and checks each `## 5.5.<i> Stage <i>` section against it (S1–S13).
130
+ **Enforced:** `validators/validate-implementation-plan-stages.py` fails a plan missing `## 5.5 Stage Map` and checks each `## 5.5.<i> Stage <i>` section against it (S1–S14).
131
131
 
132
132
  This run's final report MUST emit `## 5.5 Stage Map` and `## 5.5.<i> Stage <i>` sections per the implementation-planning profile §"Required deliverable shape". Two illustrative Stage Map tables:
133
133
 
@@ -58,6 +58,12 @@ def _run_git(args: List[str], cwd, check: bool = True) -> subprocess.CompletedPr
58
58
 
59
59
 
60
60
  def _require_eligible(stage_map, rows, stages) -> Dict[int, Dict[str, Any]]:
61
+ cancelled = sorted(
62
+ s["stage_number"] for s in stage_map
63
+ if s.get("cancelled") and s["stage_number"] in stages)
64
+ if cancelled:
65
+ raise HandoffError(
66
+ f"stages cancelled in the approved plan have no PR: {cancelled}")
61
67
  elig = {e["stage"]: e for e in compute_eligibility(stage_map, rows)}
62
68
  unknown = [n for n in stages if n not in elig]
63
69
  if unknown:
@@ -396,6 +396,8 @@ def expected_plan_item_ids(implementation_planning: Mapping[str, Any]) -> list[s
396
396
 
397
397
 
398
398
  _STARTABLE_STATUSES = frozenset({"ready", "active"})
399
+ # 다시 실행되지 않는 stage. 그 항목은 관찰만 하고 착수를 막지 않는다.
400
+ _FROZEN_STATUSES = frozenset({"done", "cancelled"})
399
401
 
400
402
  # 계획 전체를 판정하는 항목. 스테이지 하나를 고쳐도 요청하지 않은 작업이
401
403
  # 들어왔는지 다시 봐야 한다. P-Val / P-Req / P-Rb 는 여기 넣지 않는다 —
@@ -454,7 +456,7 @@ def stage_scope_bucket(
454
456
  statuses = {str(ledger.get(str(stage)) or "") for stage in stages}
455
457
  if statuses & _STARTABLE_STATUSES:
456
458
  return "in-scope"
457
- return "observed" if "done" in statuses else "deferred"
459
+ return "observed" if statuses & _FROZEN_STATUSES else "deferred"
458
460
 
459
461
 
460
462
  def _depends_on(value: object) -> tuple[int, ...]:
@@ -504,15 +506,30 @@ def _planning_stage_rows(planning: Mapping[str, Any]) -> list[tuple[int, tuple[i
504
506
  return rows
505
507
 
506
508
 
509
+ def _cancelled_stage_numbers(planning: Mapping[str, Any]) -> set[int]:
510
+ stage_map = planning.get("stageMap")
511
+ if not isinstance(stage_map, list):
512
+ return set()
513
+ return {
514
+ row["stage"] for row in stage_map
515
+ if isinstance(row, Mapping)
516
+ and isinstance(row.get("stage"), int)
517
+ and row.get("status") == "cancelled"
518
+ }
519
+
520
+
507
521
  def planning_stage_ledger(
508
522
  planning: Mapping[str, Any],
509
523
  disk_status: Mapping[str, str] | None = None,
510
524
  ) -> dict[str, str]:
511
- """지금 계획의 스테이지를 ``done`` / ``active`` / ``ready`` / ``blocked`` 로.
525
+ """지금 계획의 스테이지를 ``done`` / ``active`` / ``cancelled`` / ``ready`` /
526
+ ``blocked`` 로.
512
527
 
513
528
  디스크 원장은 이미 구현된 것만 안다. 첫 계획 run 에는 원장이 없어서
514
529
  게이트가 전 항목을 in-scope 로 읽었다. 이 함수는 현재 계획의 depends-on 으로
515
- 그 빈칸을 채운다. 디스크에 ``done`` / ``active`` 가 있으면 그쪽이 이긴다.
530
+ 그 빈칸을 채운다. 디스크에 ``done`` / ``active`` 가 있으면 그쪽이 이긴다 —
531
+ 그런 stage 를 계획이 취소했다면 그것은 검증기(S14)가 거부할 결함이고, 원장은
532
+ 실제 상태를 그대로 보여야 한다.
516
533
  """
517
534
  recorded = {
518
535
  key: value
@@ -520,6 +537,7 @@ def planning_stage_ledger(
520
537
  if value in {"done", "active", "ready", "blocked"}
521
538
  }
522
539
  done = {key for key, value in recorded.items() if value == "done"}
540
+ cancelled = _cancelled_stage_numbers(planning)
523
541
  ledger: dict[str, str] = {}
524
542
  for number, depends in _planning_stage_rows(planning):
525
543
  key = str(number)
@@ -527,6 +545,9 @@ def planning_stage_ledger(
527
545
  if status in {"done", "active"}:
528
546
  ledger[key] = status
529
547
  continue
548
+ if number in cancelled:
549
+ ledger[key] = "cancelled"
550
+ continue
530
551
  ready = all(str(dep) in done for dep in depends)
531
552
  ledger[key] = "ready" if ready else "blocked"
532
553
  return ledger
@@ -132,6 +132,16 @@ def close_stage(
132
132
  f"stage {stage} is not in this task's Stage Map (has {known or 'none'})",
133
133
  stage="stage-map",
134
134
  )
135
+ if any(
136
+ isinstance(row, dict) and row.get("stage_number") == stage
137
+ and row.get("cancelled")
138
+ for row in stage_map.stages
139
+ ):
140
+ raise StateError(
141
+ f"stage {stage} is cancelled in this task's latest plan — a cancelled "
142
+ "stage is never done; close the stage that replaces it instead",
143
+ stage="stage-map",
144
+ )
135
145
 
136
146
  recorded = last_lifecycle_status_by_stage(read_consumers(plan_run_root))
137
147
  if recorded.get(stage) in _SETTLED:
@@ -183,8 +193,9 @@ _CLI_EPILOG = r"""Usage:
183
193
  Records the `done` row an implementation run would have written, for a stage
184
194
  whose work is already committed but which never registered as done (the run
185
195
  ended before writing its carry sidecar). Refuses unless the Stage Map has that
186
- stage, no `done`/`failed` row exists for it, `--from-commit` resolves to a
187
- commit in the project repo, and the stage's conformance gate permits progress.
196
+ stage and does not mark it cancelled, no `done`/`failed` row exists for it,
197
+ `--from-commit` resolves to a commit in the project repo, and the stage's
198
+ conformance gate permits progress.
188
199
 
189
200
  Output: JSON { ok, taskKey, taskRoot, stage, headCommit, conformance,
190
201
  consumersPath }. Exit 1 on a refusal (the reason names what to do), 2 when
@@ -41,6 +41,7 @@ class StageMapStage:
41
41
  depends_on: tuple[int, ...]
42
42
  step_count: int
43
43
  exit_contract_summary: str
44
+ cancelled: bool = False
44
45
 
45
46
 
46
47
  @dataclass(frozen=True)
@@ -356,9 +357,27 @@ def _parse_data_stage_map_row(
356
357
  parse_stage_dependencies(depends_on, row_number, source_plan_path),
357
358
  step_count,
358
359
  exit_summary,
360
+ stage_row_is_cancelled(value, row_number, source_plan_path),
359
361
  )
360
362
 
361
363
 
364
+ STAGE_STATUSES = ("active", "cancelled")
365
+
366
+
367
+ def stage_row_is_cancelled(
368
+ value: dict[str, Any], row_number: int, source_plan_path: str = "",
369
+ ) -> bool:
370
+ """구조화 Stage Map 행의 `status`. 없으면 active 다."""
371
+ status = value.get("status", "active")
372
+ if status not in STAGE_STATUSES:
373
+ raise StageMapError(
374
+ "stage_map",
375
+ f"structured Stage Map row {row_number} has invalid status {status!r}",
376
+ source_plan_path,
377
+ )
378
+ return status == "cancelled"
379
+
380
+
362
381
  def _parse_schema_v2_stage_map(
363
382
  data: dict[str, Any], source_plan_path: str,
364
383
  ) -> list[StageMapStage]:
@@ -493,11 +512,17 @@ def stage_map_records(stages: Iterable[StageMapStage]) -> list[dict[str, Any]]:
493
512
  "depends_on": list(stage.depends_on),
494
513
  "step_count": stage.step_count,
495
514
  "exit_contract_summary": stage.exit_contract_summary,
515
+ "cancelled": stage.cancelled,
496
516
  }
497
517
  for stage in stages
498
518
  ]
499
519
 
500
520
 
521
+ def active_stage_records(records: Iterable[dict[str, Any]]) -> list[dict[str, Any]]:
522
+ """취소되지 않은 stage 기록 — 통합·whole-task 게이트·handoff 가 순회하는 집합."""
523
+ return [record for record in records if not record.get("cancelled")]
524
+
525
+
501
526
  def _latest_plan_report(task_root: Path) -> Path | None:
502
527
  reports_dir = RunRef.from_task_root(
503
528
  task_root, "implementation-planning"
@@ -30,8 +30,9 @@ _CLI_EPILOG = r"""Usage:
30
30
 
31
31
  Output: default and --json emit JSON { ok, taskKey, taskRoot, state, sourcePlanPath,
32
32
  stages:[…], doneStages:[int], planning:{…} }; --text emits fixed count/name/value fields.
33
- Each stage row carries stage_number, title, depends_on, step_count and exit_contract_summary; a
34
- schema-v2 report adds that stage's sliceValue, acceptance and exitContract.
33
+ Each stage row carries stage_number, title, depends_on, step_count, exit_contract_summary and
34
+ cancelled (true for a row the plan marks `status: cancelled`); a schema-v2 report adds that
35
+ stage's sliceValue, acceptance and exitContract.
35
36
  planning carries the report's task-level rows — rollbackStrategy, validationChecklist,
36
37
  crossProjectDependencies, dependencyMigrationRisk, recommendedOption — so a consumer renders the
37
38
  plan instead of re-summarising the report body. Both are absent for a schema-v1 report.
@@ -53,10 +53,12 @@ class StageLifecycle:
53
53
  step_count: int
54
54
  deps_satisfied: bool
55
55
  blocked_by: list[int]
56
+ cancelled: bool = False
56
57
 
57
58
  @property
58
59
  def status(self) -> str:
59
- """Lifecycle state, in precedence order: done > active > ready > blocked.
60
+ """Lifecycle state, in precedence order:
61
+ done > cancelled > active > ready > blocked.
60
62
 
61
63
  ``started`` and ``reserved`` collapse into ``active`` because every
62
64
  caller that distinguishes them already words the two the same way (see
@@ -64,6 +66,8 @@ class StageLifecycle:
64
66
  """
65
67
  if self.done:
66
68
  return "done"
69
+ if self.cancelled:
70
+ return "cancelled"
67
71
  if self.started or self.reserved:
68
72
  return "active"
69
73
  return "ready" if self.deps_satisfied else "blocked"
@@ -171,6 +175,11 @@ class StageLifecycleSnapshot:
171
175
  raise StageTargetError(
172
176
  f"--stage {n} already completed (consumers.jsonl status:done exists)"
173
177
  )
178
+ if lifecycle.status == "cancelled":
179
+ raise StageTargetError(
180
+ f"--stage {n} is cancelled in the approved plan's Stage Map; "
181
+ "it is never executed — run the stage that replaces it"
182
+ )
174
183
  if lifecycle.status == "active":
175
184
  # Wording is load-bearing: okstra-run's unattended chain matches it
176
185
  # to terminate normally instead of raising an exception gate.
@@ -183,8 +192,8 @@ class StageLifecycleSnapshot:
183
192
  blocked = [lc for lc in self.lifecycles if lc.status == "blocked"]
184
193
  if not blocked:
185
194
  return (
186
- "no stage is ready: every stage is done or occupied by "
187
- "another run"
195
+ "no stage is ready: every stage is done, cancelled, or "
196
+ "occupied by another run"
188
197
  )
189
198
  detail = "; ".join(
190
199
  f"stage {lc.stage} blocked by "
@@ -203,6 +212,7 @@ class StageLifecycleSnapshot:
203
212
  return [
204
213
  lifecycle.handoff_eligibility_record()
205
214
  for lifecycle in self.lifecycles
215
+ if lifecycle.status != "cancelled"
206
216
  ]
207
217
 
208
218
 
@@ -242,6 +252,7 @@ def stage_lifecycle_snapshot_from_state(
242
252
  pr_covered=stage_number in consumer_state.pr_covered_stages,
243
253
  head_commit=str(done_row.get("head_commit") or ""),
244
254
  report_path=str(done_row.get("report_path") or ""),
255
+ cancelled=bool(stage.get("cancelled")),
245
256
  ))
246
257
  return StageLifecycleSnapshot(
247
258
  stage_map=stage_map,
@@ -710,8 +721,11 @@ def _resolve_and_integrate_whole_task_unlocked(
710
721
  """Integrate and resolve a whole-task target while the caller owns the lock."""
711
722
  from .consumers import latest_done_by_stage
712
723
  from .stage_integrate import IntegrateError, integrate_stages
724
+ from .stage_map import active_stage_records
713
725
  from .worktree import _git, is_dirty_excluding_okstra
714
726
 
727
+ stage_map = active_stage_records(stage_map)
728
+
715
729
  # 끝나지 않은 stage 가 있으면 _resolve_whole_task_target 이 어차피 거부한다.
716
730
  # 그 전에 끝난 stage 를 머지해 두면 거부된 요청이 task worktree 에 머지를
717
731
  # 남긴다(되돌리지 않는다). 머지하기 전에 같은 조건으로 거부한다.
@@ -231,7 +231,8 @@ def _stage_lifecycle_snapshot(
231
231
  return read_stage_lifecycle_snapshot(
232
232
  [{"stage_number": s.stage_number,
233
233
  "depends_on": list(s.depends_on),
234
- "step_count": s.step_count} for s in stages],
234
+ "step_count": s.step_count,
235
+ "cancelled": s.cancelled} for s in stages],
235
236
  Path(state.approved_plan_path).resolve().parents[1],
236
237
  recover_from_carry=False,
237
238
  reserved_stages=reserved_stages,
@@ -249,7 +249,9 @@ def _whole_task_allowed(
249
249
  if done is None:
250
250
  done = _stage_lifecycle_snapshot(state, stages).done_stages
251
251
  return whole_task_verification_allowed(
252
- stage_numbers=tuple(stage.stage_number for stage in stages),
252
+ stage_numbers=tuple(
253
+ stage.stage_number for stage in stages if not stage.cancelled
254
+ ),
253
255
  done_stages=done,
254
256
  )
255
257
 
@@ -468,7 +470,7 @@ def _submit_impl_stage_pick(state: WizardState, answer: str) -> Optional[str]:
468
470
  state, stages, reserved_stages=_reserved_stage_numbers(state))
469
471
  done = snapshot.done_stages
470
472
  occupied = {lc.stage for lc in snapshot.lifecycles
471
- if lc.status in ("done", "active")}
473
+ if lc.status in ("done", "active", "cancelled")}
472
474
  chosen = _impl_chosen_stages(t, picks, answer, all_nums, occupied)
473
475
  ordered = order_stage_closure(
474
476
  [(s.stage_number, s.depends_on) for s in stages], chosen, done)
@@ -101,8 +101,10 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
101
101
  `qa/scripts/`, plus the manifest entry naming them
102
102
  (`_implementation-executor.md` §"Stage conformance script", §"Real-IO test
103
103
  isolation"), plus `qa/output/` for whatever those scripts write when the
104
- executor runs them against its own work — a baseline capture, build logs
105
- (§"Verifier gates are not yours to run" lets it run them). Everything else
104
+ executor runs them against its own work — build logs, measurements
105
+ (§"Verifier gates are not yours to run" lets it run them) — and
106
+ `qa/baseline/` for a captured baseline or `--compare` target, which the
107
+ final verifier must not rewrite. Everything else
106
108
  under `qa/` — the self-mock sidecar and its diff, the conformance run's
107
109
  `result-*.json` — is written while the verifier runs its own gates.
108
110
  Keeping the executor's grant to those three entries is
@@ -124,6 +126,7 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
124
126
  return (
125
127
  task_qa_dir(task_root) / "scripts",
126
128
  task_qa_dir(task_root) / "output",
129
+ task_qa_dir(task_root) / "baseline",
127
130
  task_conformance_manifest_file(task_root),
128
131
  )
129
132
  if role == "verifier":
@@ -154,7 +157,9 @@ def _inherited_qa_evidence(
154
157
  baseline overwrote the implementation's RED evidence and still passed the
155
158
  audit (observed 2026-09-26, dev-11054 `qa/baseline/base.json`). New files and
156
159
  the conformance results it re-runs (`qa/result-<stageKey>.json`, the only qa
157
- write `final_verification/boundary.json` names) stay writable.
160
+ write `final_verification/boundary.json` names) stay writable, and so does
161
+ `qa/output/`: the Tier 3 scripts it must re-run rewrite their own outputs
162
+ there (dev-11127 final-verification); baselines live in `qa/baseline/`.
158
163
  """
159
164
  if role != "verifier" or task_type != "final-verification" or task_root is None:
160
165
  return ()
@@ -165,9 +170,11 @@ def _inherited_qa_evidence(
165
170
  conformance_result_file(qa_dir, key)
166
171
  for key in _conformance_stage_keys(task_root)
167
172
  }
173
+ regenerated = qa_dir / "output"
168
174
  return tuple(sorted(
169
175
  path for path in qa_dir.rglob("*")
170
176
  if path.is_file() and not path.is_symlink() and path not in rerun
177
+ and not path.is_relative_to(regenerated)
171
178
  ))
172
179
 
173
180
 
@@ -8158,6 +8158,13 @@
8158
8158
  "exitContractSummary": {
8159
8159
  "type": "string",
8160
8160
  "minLength": 1
8161
+ },
8162
+ "status": {
8163
+ "enum": [
8164
+ "active",
8165
+ "cancelled"
8166
+ ],
8167
+ "description": "Absent means active. A cancelled stage keeps its row and number so the Stage Map still runs 1..N; execution, gates and handoff skip it. Only a stage that has not started may be cancelled."
8161
8168
  }
8162
8169
  }
8163
8170
  },
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env python3
2
- """S1–S13 checks for the Stage Map structure of an approved
2
+ """S1–S14 checks for the Stage Map structure of an approved
3
3
  implementation-planning final-report.md. Run from prepare_task_bundle
4
4
  of `implementation` task or standalone."""
5
5
 
@@ -35,6 +35,7 @@ from okstra_ctl.stage_map import ( # noqa: E402
35
35
  parse_stage_map_text,
36
36
  schema_v2_report,
37
37
  stage_number_gap_message,
38
+ stage_row_is_cancelled,
38
39
  )
39
40
 
40
41
  HARD_STEP_CAP = 8
@@ -68,9 +69,14 @@ BARE_GIT_STATUS = re.compile(r"\bgit\b[^&|;]*?\bstatus\b[^&|;]*?--(?:porcelain|s
68
69
  CLEAN_GATE_COMMAND = "okstra worktree-status --check-clean"
69
70
 
70
71
 
72
+ # S14 — 취소 stage 가 될 수 없는 상태. 이미 착수했거나 끝난 stage 는 통합 기준선의
73
+ # 일부라 취소하면 그 위에 쌓인 브랜치와 갈라진다(ADR-0015); 고치려면 max+1 보정 stage.
74
+ _UNCANCELLABLE_STATUSES = ("done", "active")
75
+
76
+
71
77
  @dataclass
72
78
  class ValidationError:
73
- code: str # S1..S13
79
+ code: str # S1..S14
74
80
  stage: int # 0 = global
75
81
  message: str
76
82
 
@@ -380,13 +386,19 @@ def _check_conformance_declaration(
380
386
  return errs
381
387
 
382
388
 
383
- def _check_depends_on(stages: List[StageMapStage]) -> List[ValidationError]:
389
+ def _check_depends_on(
390
+ stages: List[StageMapStage], cancelled: frozenset[int] = frozenset(),
391
+ ) -> List[ValidationError]:
384
392
  errs: List[ValidationError] = []
385
393
  valid = {s.stage_number for s in stages}
386
394
  for s in stages:
387
395
  for d in s.depends_on:
388
396
  if d == s.stage_number:
389
397
  errs.append(ValidationError("S8", s.stage_number, "self depends-on"))
398
+ elif d in cancelled:
399
+ errs.append(ValidationError("S14", s.stage_number,
400
+ f"depends-on {d}, which this plan cancels — a cancelled stage "
401
+ "is never executed; depend on the stage that replaces it"))
390
402
  elif d not in valid:
391
403
  errs.append(ValidationError("S6", s.stage_number,
392
404
  f"depends-on {d} does not exist"))
@@ -495,6 +507,7 @@ def _data_stage_metas(
495
507
  depends = parse_stage_dependencies(
496
508
  str(row.get("dependsOn") or ""), row_number, ""
497
509
  )
510
+ cancelled = stage_row_is_cancelled(row, row_number)
498
511
  except StageMapError as exc:
499
512
  errors.append(ValidationError("S2", row["stage"], exc.reason))
500
513
  continue
@@ -504,6 +517,7 @@ def _data_stage_metas(
504
517
  depends,
505
518
  row.get("stepCount") if isinstance(row.get("stepCount"), int) else -1,
506
519
  str(row.get("exitContractSummary") or ""),
520
+ cancelled,
507
521
  ))
508
522
  return rows, errors or _stage_numbers_monotonic(rows)
509
523
 
@@ -670,6 +684,28 @@ def _check_data_stage_identities(
670
684
  return errs
671
685
 
672
686
 
687
+ def _check_cancelled_stage_ledger(
688
+ stage_map: List[StageMapStage], planning: dict,
689
+ ) -> List[ValidationError]:
690
+ """S14: 원장이 이미 착수했거나 끝났다고 기록한 stage 는 취소할 수 없다."""
691
+ verification = planning.get("planBodyVerification")
692
+ ledger = (
693
+ verification.get("stageLedger") if isinstance(verification, dict) else None
694
+ )
695
+ if not isinstance(ledger, dict):
696
+ return []
697
+ return [
698
+ ValidationError("S14", meta.stage_number,
699
+ f"stage {meta.stage_number} is cancelled but the Stage Ledger records "
700
+ f"it {ledger[str(meta.stage_number)]!r}; only a ready or blocked stage "
701
+ "may be cancelled — keep it and add a correction stage numbered after "
702
+ "the highest stage instead")
703
+ for meta in stage_map
704
+ if meta.cancelled
705
+ and ledger.get(str(meta.stage_number)) in _UNCANCELLABLE_STATUSES
706
+ ]
707
+
708
+
673
709
  def collect_data_validation_errors(
674
710
  planning: dict, user_bypassed: frozenset[int] = frozenset(),
675
711
  ) -> List[ValidationError]:
@@ -703,14 +739,19 @@ def collect_data_validation_errors(
703
739
  if identity_errors:
704
740
  return errors
705
741
  errors.extend(_check_data_step_counts(stage_map, stages))
706
- errors.extend(_check_depends_on(stage_map))
742
+ # 번호(S2)·대응(S3)·스텝 수는 모든 행에, 의존(S6/S8)·병렬 안전(S9)은 실행될
743
+ # 행에만 건다 — 취소 stage 와 그 대체 stage 는 같은 파일을 예측하기 마련이다.
744
+ active = [meta for meta in stage_map if not meta.cancelled]
745
+ cancelled = frozenset(meta.stage_number for meta in stage_map if meta.cancelled)
746
+ errors.extend(_check_depends_on(active, cancelled))
747
+ errors.extend(_check_cancelled_stage_ledger(stage_map, planning))
707
748
  errors.extend(_report_shared_parallel_files({
708
749
  meta.stage_number: set(PATH_TOKEN.findall(
709
750
  str((next(
710
751
  (s for s in stages if s.get("stage") == meta.stage_number), {}
711
752
  )).get("exitContract") or "")
712
753
  ))
713
- for meta in stage_map
754
+ for meta in active
714
755
  if not meta.depends_on
715
756
  }))
716
757
  for stage in stages: