okstra 0.196.0 → 0.197.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/dist/cli-registry.mjs +6 -0
  2. package/dist/cli-registry.mjs.map +1 -1
  3. package/docs/cli.md +14 -1
  4. package/docs/project-structure-overview.md +1 -0
  5. package/package.json +1 -1
  6. package/runtime/BUILD.json +2 -2
  7. package/runtime/prompts/host-orchestration/implementation-planning.md +56 -0
  8. package/runtime/prompts/lead/plan-body-verification.md +21 -3
  9. package/runtime/prompts/profiles/implementation-planning.md +3 -2
  10. package/runtime/prompts/wizard/prompts.ko.json +2 -2
  11. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +15 -0
  12. package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +15 -0
  13. package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +15 -0
  14. package/runtime/python/okstra_ctl/agent/prompt_cli/dynamic_verifier.py +12 -10
  15. package/runtime/python/okstra_ctl/blocking_checks.py +19 -0
  16. package/runtime/python/okstra_ctl/conformance.py +36 -5
  17. package/runtime/python/okstra_ctl/dispatch_core.py +11 -6
  18. package/runtime/python/okstra_ctl/dispatch_state.py +11 -7
  19. package/runtime/python/okstra_ctl/domain/worker_stream.py +4 -5
  20. package/runtime/python/okstra_ctl/final_report_schema.py +62 -1
  21. package/runtime/python/okstra_ctl/group_context.py +96 -3
  22. package/runtime/python/okstra_ctl/plan_items.py +14 -0
  23. package/runtime/python/okstra_ctl/plan_items_cli.py +45 -3
  24. package/runtime/python/okstra_ctl/run.py +77 -0
  25. package/runtime/python/okstra_ctl/session_transcript.py +4 -4
  26. package/runtime/python/okstra_ctl/set_work_status.py +30 -1
  27. package/runtime/python/okstra_ctl/stage_close.py +244 -0
  28. package/runtime/python/okstra_ctl/tdd_bypass.py +131 -0
  29. package/runtime/python/okstra_ctl/wizard/engine.py +27 -24
  30. package/runtime/python/okstra_ctl/wizard/picker_navigation.py +37 -14
  31. package/runtime/python/okstra_ctl/wizard/roles.py +16 -11
  32. package/runtime/python/okstra_ctl/worker_prompt_contract.py +15 -1
  33. package/runtime/python/okstra_ctl/worker_prompt_policy.py +45 -2
  34. package/runtime/skills/okstra-brief-gen/SKILL.md +24 -7
  35. package/runtime/skills/okstra-inspect/facets/recap.md +17 -1
  36. package/runtime/skills/okstra-run/SKILL.md +63 -4
  37. package/runtime/templates/reports/group-context.template.md +1 -1
  38. package/runtime/validators/validate-implementation-plan-stages.py +109 -23
  39. package/runtime/validators/validate-run.py +63 -8
@@ -83,16 +83,59 @@ WORKER_ERROR_CONTRACT_FILENAME = "worker-error-contract.md"
83
83
  # 따르되 라운드 원장 밖이라 번호가 없다. 실측(2026-09-09 dev-10642): 이 kind 가
84
84
  # 없어 gap 검증 프롬프트를 만들 수 없었고 gap 3건 전부 `gapsUnverified` 로 남았다.
85
85
  CRITIC_VERIFY_DISPATCH_KIND = "critic-verify"
86
+ REVERIFY_DISPATCH_KIND_PREFIX = "reverify-r"
87
+ # 계획 본문 검증(§5.5.9) 라운드의 dispatch kind. 결과 파일명 규약이 이미
88
+ # `<role>-worker-plan-verify-r<N>-…` 이라(`plan-body-verification.md` §"Round
89
+ # protocol", `dispatch_state._plan_verify_result_workers`) kind 도 같은 번호를
90
+ # 단다. 실측(2026-09-09 dev-10642 implementation-planning 001): 이 kind 가 없어
91
+ # 계획 항목 큐를 `reverify-r<N>` 으로 보낼 수밖에 없었고, 그러면 검증 계약이
92
+ # `okstra convergence reverify-prompt` 서명을 요구하는데 계획 항목 큐의 정본
93
+ # 렌더러는 `okstra plan-items prompt` 라 통과할 값이 하나도 없었다. 라운드가
94
+ # 한 번도 열리지 못한 채 `planBodyVerification.roundCount` 가 0 으로 남았다.
95
+ PLAN_VERIFY_DISPATCH_KIND_PREFIX = "plan-verify-r"
96
+
97
+
98
+ def _numbered_round(dispatch_kind: str, prefix: str) -> int | None:
99
+ """`<prefix><N>` 의 N. 접두사가 다르거나 N 이 양의 정수가 아니면 None."""
100
+ if not dispatch_kind.startswith(prefix):
101
+ return None
102
+ suffix = dispatch_kind[len(prefix):]
103
+ if not suffix.isdigit() or int(suffix) < 1:
104
+ return None
105
+ return int(suffix)
86
106
 
87
107
 
88
108
  def is_verification_dispatch_kind(dispatch_kind: str) -> bool:
89
- """번호 reverify 라운드와 critic gap 검증 — 검증 프롬프트 계약을 받는 kind."""
109
+ """번호 reverify·계획 본문 라운드와 critic gap 검증 — 검증 프롬프트 계약을
110
+ 받는 kind."""
90
111
  return (
91
- dispatch_kind.startswith("reverify-r")
112
+ dispatch_kind.startswith(REVERIFY_DISPATCH_KIND_PREFIX)
113
+ or dispatch_kind.startswith(PLAN_VERIFY_DISPATCH_KIND_PREFIX)
92
114
  or dispatch_kind == CRITIC_VERIFY_DISPATCH_KIND
93
115
  )
94
116
 
95
117
 
118
+ def is_plan_verify_dispatch_kind(dispatch_kind: str) -> bool:
119
+ """계획 본문 검증 라운드인가. 검증 계약이 요구하는 렌더러 서명이 번호
120
+ reverify 와 다르므로 계약 검사가 이 둘을 갈라야 한다."""
121
+ return _numbered_round(dispatch_kind, PLAN_VERIFY_DISPATCH_KIND_PREFIX) is not None
122
+
123
+
124
+ def verification_dispatch_round(dispatch_kind: str) -> int | None:
125
+ """검증 kind 가 적는 라운드 번호. 번호 없는 kind(`critic-verify`)는 1,
126
+ 검증 kind 가 아니거나 번호가 깨졌으면 None.
127
+
128
+ 예약(`agent-prompt materialize`)과 디스패치가 같은 값을 적어야 한다 —
129
+ validate-run 이 team-state 의 kind 와 예약된 invocation 의 `dispatchKind`
130
+ 를 대조하므로, 두 계산이 갈리면 그 디스패치가 거부된다."""
131
+ if dispatch_kind == CRITIC_VERIFY_DISPATCH_KIND:
132
+ return 1
133
+ for prefix in (REVERIFY_DISPATCH_KIND_PREFIX, PLAN_VERIFY_DISPATCH_KIND_PREFIX):
134
+ if dispatch_kind.startswith(prefix):
135
+ return _numbered_round(dispatch_kind, prefix)
136
+ return None
137
+
138
+
96
139
  @dataclass(frozen=True)
97
140
  class PromptPlan:
98
141
  audience: PromptAudience
@@ -1024,13 +1024,30 @@ per-page ratio instead of origin load). If the file is absent, ask once via
1024
1024
  - `Skip — this group needs no shared context`.
1025
1025
 
1026
1026
  Never fill the skeleton yourself from the tickets: a ticket summary decides
1027
- nothing, and the sections ask for what the tickets do not say. If the file
1028
- exists, leave it untouched and mention its path in the hand-off block. The
1029
- file's trailing `## Task Memory` region (between `<!-- okstra:task-memory:begin -->`
1030
- and `end`) is okstra's: `report-finalize` rewrites it after every run with the
1031
- group's start order and each task's latest conclusion, and a group whose
1032
- runs finished before anyone created the file already has one holding that
1033
- region alone — `init` then inserts the human sections above it.
1027
+ nothing, and the sections ask for what the tickets do not say.
1028
+
1029
+ If the file exists, what you do with it depends on what you are writing:
1030
+
1031
+ - **A new brief** leaves it untouched. Mention its path in the hand-off block.
1032
+ - **A rewrite or correction of an existing brief** reconciles it in the same
1033
+ response, whenever the change supersedes something the group document
1034
+ states — a Definition of Better figure, a success signal, a Group-Wide
1035
+ Constraint, a Ticket Relation, or what it says is executable from the
1036
+ project root. okstra copies this file into every run's
1037
+ `instruction-set/task-group-context.md` and carries it in the analysis
1038
+ packet **ahead of the brief extract**, so a sentence you corrected in the
1039
+ brief is still outranked by the group document's stale copy of it. Qualify a
1040
+ diverged figure with the date and method of the measurement that contradicts
1041
+ it rather than deleting it, and name the brief the correction came from.
1042
+
1043
+ Either way, edit only the authored sections. The file's trailing `## Task Memory`
1044
+ region (between `<!-- okstra:task-memory:begin -->` and `end`) is okstra's: it
1045
+ holds the group's start order and each task's latest conclusion, and okstra
1046
+ redraws it after every run's `report-finalize`, at `okstra set-work-status`, and
1047
+ before each run copies the file into its instruction set. Anything written there
1048
+ by hand is overwritten without warning. A group whose runs finished before
1049
+ anyone created the file already has one holding that region alone — `init` then
1050
+ inserts the human sections above it.
1034
1051
 
1035
1052
  Then stop. Do not invoke `okstra-run` directly — the user chooses when to
1036
1053
  proceed, and they may want to edit the brief externally first. In the
@@ -6,7 +6,7 @@ Loaded lazily by the dispatch table in `SKILL.md` (core). Shared rules — Step
6
6
 
7
7
  Trigger phrases: "okstra recap", "recap", "work summary", "summarize this task", "before/after summary", "explain this work", "task question".
8
8
 
9
- On top of the `.okstra` artifacts accumulated for a single task-id — or for every task of one task-group — (a) produce a before/after summary and (b) answer free-form questions about that work. By default it reads only the `.okstra/` subtree (artifact mode). It expands to code mode only when the user explicitly asks to look at the code changes too. This sub-command performs only the `recap-log.jsonl` append and the `notes/` note authoring (recap.5); it never mutates `task-manifest.json` / catalog / timeline / `group-context.md`.
9
+ On top of the `.okstra` artifacts accumulated for a single task-id — or for every task of one task-group — (a) produce a before/after summary and (b) answer free-form questions about that work. By default it reads only the `.okstra/` subtree (artifact mode). It expands to code mode only when the user explicitly asks to look at the code changes too. This sub-command performs the `recap-log.jsonl` append, the `notes/` note authoring (recap.5), and the group-context reconciliation that note triggers (recap.6); it never mutates `task-manifest.json` / catalog / timeline, and it touches `group-context.md` only in the authored sections above the `<!-- okstra:task-memory:begin -->` marker.
10
10
 
11
11
  ### recap.1 — Resolve target
12
12
 
@@ -144,3 +144,19 @@ Write the body to the scratchpad as markdown first, then pass it with `--body-fi
144
144
  4. **`notes/` is inert to okstra** — no run reads it automatically. After writing, relay the `clarificationResponseArg` the CLI printed (e.g. `--clarification-response <notePath>`) to the user verbatim, telling them it only takes effect when the next run is executed with that argument.
145
145
 
146
146
  **Guardrail:** `.okstra/` is gitignored — treat `notes/` as local scratch and never `git add` it. Creating a new note is easy to undo (delete the file), so always prefer it over editing a generated or user-owned file.
147
+
148
+ ### recap.6 — Reconcile the task-group context (required whenever recap.5 writes a note)
149
+
150
+ A group's shared context does not stay true while its tasks run. Figures get re-measured, a success signal turns out to be satisfiable without a fix, a constraint names the wrong place. `group-context.md` states in its own header that okstra copies it into each run's `instruction-set/task-group-context.md` and carries it in the analysis packet's `## Task-Group Context` section, **ahead of the brief extract** — so a stale sentence there outranks a corrected brief for every task in the group.
151
+
152
+ **After writing a note, read `<PROJECT_ROOT>/.okstra/briefs/<task-group>/group-context.md` if it exists.** If anything the note establishes touches its Definition of Better figures, success signals, Group-Wide Constraints, Ticket Relations, or its statement of what is executable from the project root, update those sections **in the same response that wrote the note**. A note written without that check is an incomplete deliverable, the same way a `.project-docs/` document without its index row is.
153
+
154
+ Three rules for the edit:
155
+
156
+ 1. **Only above the marker.** Edit the authored sections above `<!-- okstra:task-memory:begin -->`. The region below is okstra's projection — start order, per-task status, recorded runs — and it is redrawn from the task manifests at every `report-finalize`, at `okstra set-work-status`, and before each run copies the file. Anything you write there is overwritten without warning.
157
+ 2. **Qualify a diverged figure; do not delete it.** An audit number is the baseline for the window it was measured in, not a current reading. Where a direct measurement contradicts it, say so with the measurement's date and method, and keep both. Deleting the old figure destroys the reason the group exists.
158
+ 3. **Say which task and note the correction came from.** The next reader needs to get from the sentence back to its evidence.
159
+
160
+ If the group has no `group-context.md`, do not create one here — `okstra group-context init` (via okstra-brief-gen) owns that skeleton, and creating an empty one pre-empts the sections a human meant to author.
161
+
162
+ The same exposure applies to any other edit that supersedes group-wide facts, a brief rewrite most of all. Treat this check as belonging to the fact, not to this sub-command.
@@ -53,7 +53,7 @@ The wizard tells you which relay operation to use via `next.interaction.kind`. S
53
53
  - `kind: "done"` → input collection finished; move to Step 5.
54
54
  - `kind: "aborted"` → the user picked abort; the wizard is terminally cancelled. Tell the user on one short line that the run setup was aborted, delete the state file (`rm` with the literal path), and stop this skill — do NOT call `render-args` or `render-bundle` (the wizard rejects `render-args` on an aborted state).
55
55
 
56
- When native single selection is available, the runtime divides long lists and unsupported multi-selections into selectable screens. Render the returned screen without rebuilding the full list. This applies to plan confirmation, stages, role counts, and provider/model lists. Submit navigation and completion option values normally; the runtime retains the current step until its answer is complete. A ban on textual choice lists does not authorize asking the user to type a model identifier. If a required selector is unavailable, preserve state and follow the relay's recovery. Genuine text steps still collect text.
56
+ When native single selection is available, the runtime divides long lists and unsupported multi-selections into selectable screens. Render the returned screen without rebuilding the full list. This applies to plan confirmation, stages, role counts, and provider/model lists. Submit navigation and completion option values normally; the runtime retains the current step until its answer is complete. A ban on textual choice lists does not authorize asking the user to type a model identifier. If a required selector is unavailable, preserve state and follow the relay's `recovery` object. Genuine text steps still collect text.
57
57
 
58
58
  Submit the answer shape required by `interaction.answerProtocol`; do not add normalization beyond the registered relay's explicit mapping. Invalid, out-of-range, or ambiguous answers return `ok: false` and must re-render the same complete interaction.
59
59
 
@@ -83,11 +83,11 @@ On `Okstra preflight: ready`, require `Runtime readiness: ready` before Step 2.
83
83
 
84
84
  Carry the fixed `Lead entry mode` line into Step 2's `--entry-mode` as a literal. It is the mode the run must launch in, which is not always the mode this session is in: a host adapter may report that this session cannot host the lead while the run itself is ready. `spawn-process` means okstra starts its own lead process and this session is not the lead — when the line reads `spawn-process` and a readiness check carries `action: spawn-unsandboxed-codex-lead`, tell the user in one line that their current Codex session is sandboxed so okstra will open an unsandboxed lead instead, then continue. Never substitute `current-session` for a `spawn-process` answer, and never recompute the value from the host ID.
85
85
 
86
- For a ready response, read the absolute path in the fixed `Relay contract` line with the current host's file-read primitive. Do not derive the path from the host ID or search `PATH`. In that file, find the `Wizard interaction relay` JSON block, require `schemaVersion: 1` and `runtime` equal to the fixed `Runtime` line, then take its `semanticFunctions` allowlist and intersect it with the functions the live harness exposes. The live harness does not expose tools named `native_single_select`. Map each allowlist token to the matching `interactions` kind (`native_single_select` → `native-single`, `native_multi_select` → `native-multi`, `native_question_group` → `native-group`). Include the token in the intersection only when this session can call the string in that kind's `function` field. If the kind is absent from `interactions`, omit the token. Pass only that intersection to Step 2; `plain_text_input` must be present. Keep the parsed `interactions` object for Step 3's function/input/response conversion. An unreadable file, malformed block, runtime mismatch, absent `plain_text_input`, or later interaction kind missing from the object is a host relay contract failure: show the problem and stop rather than guessing.
86
+ For a ready response, read the absolute path in the fixed `Relay contract` line with the current host's file-read primitive. Do not derive the path from the host ID or search `PATH`. In that file, find the `Wizard interaction relay` JSON block, require `schemaVersion: 1` and `runtime` equal to the fixed `Runtime` line, then take its `semanticFunctions` allowlist and intersect it with the functions the live harness exposes. The live harness does not expose tools named `native_single_select`. Map each allowlist token to the matching `interactions` kind (`native_single_select` → `native-single`, `native_multi_select` → `native-multi`, `native_question_group` → `native-group`). Include the token in the intersection only when this session can call the string in that kind's `function` field. If the kind is absent from `interactions`, omit the token. Pass only that intersection to Step 2; `plain_text_input` must be present. Keep the parsed `interactions` object for Step 3's function/input/response conversion, and the parsed `recovery` object for the branch below. An unreadable file, malformed block, runtime mismatch, absent `plain_text_input`, or later interaction kind missing from the object is a host relay contract failure: show the problem and stop rather than guessing.
87
87
 
88
88
  If the successful fixed projection has `Relay contract: -`, enter the compatibility branch below. In that branch only, declare `plain_text_input` and keep its built-in `interactions` mapping for Step 3; do not assume a native tool from the host ID. The existing `unknown command: preflight` branch remains the authoritative stale-CLI failure.
89
89
 
90
- Before calculating the effective intersection, apply the registered relay's client-specific tool selection and live mode restrictions. A callable question tool does not necessarily render a selector in the current client. In Codex, use the synchronous picker when permitted; asynchronous question cards are a desktop fallback, not a terminal picker. If the user requires a selectable interface and the current client cannot provide it, preserve the wizard state and follow the relay's recovery instead of printing the option list.
90
+ Before calculating the effective intersection, apply the registered relay's client-specific tool selection and live mode restrictions. A callable question tool does not necessarily render a selector in the current client. In Codex, use the synchronous picker when permitted; asynchronous question cards are a desktop fallback, not a terminal picker. A declared native function can still fail at call time: the client refuses the call, the session has no view to present it in, or it returns no answer. That is not a reason to stop, and not a reason to try a different native function. Follow the relay's `recovery.native-question-refused` entry the first time it happens — drop the tokens it names from the intersection, keep `plain_text_input`, tell the user in one line that the host picker is unavailable, and render every remaining screen through the text mapping. Numbered text is that entry's own recovery path, so the ban on printing a numbered list does not apply once it fires. The wizard state file is untouched by a refused call: `okstra wizard step --state-file <path> --no-submit` returns the pending prompt again.
91
91
 
92
92
  Plan adoption (`approve_plan_confirm`) is a workflow choice: present the wizard's existing options using the same selector as other `pick` steps. Follow the relay's distinction between plan decisions and execution permissions; the word "approval" alone is not a reason to replace a selector with a typed confirmation.
93
93
 
@@ -198,7 +198,7 @@ That is the entire interactive flow. The wizard handles:
198
198
  - base-ref pick + git rev-parse validation (skipped when reusing an active worktree),
199
199
  - `implementation`-only sub-flow: approved-plan path (frontmatter `approved: true` check) + stage pick (`auto` = the earliest incomplete stage whose dependencies are satisfied, or a specific stage number). Implementer slots use role-count / role-model like every other role (`executor` is only a compatibility alias for `implementer`). When an approved plan is selected and a `## PLAN DECISION` sidecar carrying `Status: approved`, exported from the report — matching the plan on source-report·seq — is detected in that run's sibling `user-responses/`, the approve-confirm step expands to 3 options (`yes_apply` recommended: approve + apply the option as exported / `yes` approve only / `no` abort) — `yes_apply` validates the option against the plan's `optionCandidates` before applying it via the existing approval·option path,
200
200
  - `release-handoff`-only sub-flow: after the approved plan auto-resolves, a `handoff_stage_pick` multi-select — choose an eligible stage bundle (stage-group) or the whole task (when an accepted whole-task verification report exists); the result goes out as render-args' `stages` key (csv, empty when whole-task),
201
- - launch selection after identity/worktree steps: one screen per static role, in profile order. A role that can run several instances (`max > 1`) is a checkbox step `role-models:<role>` (`multi: true`) — the number of models checked is the number of instances, there is no separate count question; the label states the profile range and recommended count, every executable candidate is listed on that one screen (defaults first, the recommended set flagged), and when the list exceeds the host's native checkbox limit the runtime either splits it into several checkbox questions on one `pick_group` screen (interaction plan `native-group`; the question steps are `role-models:<role>#1`, `#2`, … and the answer is one JSON object keyed by them, each value a CSV) when the host's native question group holds every option, or returns the interaction plan `numbered-multi` — render the whole list, never a shortlist or pages. An optional role (`min = 0`, e.g. critic) carries a `추가 안 함` row. A fixed single role (`min = max = 1`, e.g. report-writer) is a single pick `role-model:<role>:1`. current-session lead is this session and is listed on the confirmation summary, not as a wizard step. The wizard does not fork on defaults-vs-customize, does not show a provider roster multi-pick, and does not offer a separate implementer-provider pick. Dynamic verifiers are not chosen at launch. `--workers` is compatibility-only, not a launch picker. Repeated `--role-count` / `--role-model` tokens on `renderArgv` are intentional,
201
+ - launch selection after identity/worktree steps: one screen per static role, in profile order. A role that can run several instances (`max > 1`) is a checkbox step `role-models:<role>` (`multi: true`) — the number of models checked is the number of instances, there is no separate count question; the label states the profile range and recommended count, every executable candidate is listed on that one screen (defaults first, the recommended set flagged), and when the list exceeds the host's native checkbox limit the runtime either splits it into several checkbox questions on one `pick_group` screen (interaction plan `native-group`; the question steps are `role-models:<role>#1`, `#2`, … and the answer is one JSON object keyed by them, each value a CSV) when the host's native question group holds every option, or returns the interaction plan `numbered-multi` — render the whole list, never a shortlist or pages. An optional role (`min = 0`, e.g. critic) carries a `추가 안 함` row. A fixed single role (`min = max = 1`, e.g. report-writer) is a single pick `role-model:<role>:1`, and a role capped at one model (`max = 1`, e.g. critic) is a single pick on its `role-models:<role>` step; both split the same way when they exceed the native option limit — the tabs are checkbox questions, the answer is the same keyed JSON object, and the wizard rejects a screen whose tabs together name more than one value. current-session lead is this session and is listed on the confirmation summary, not as a wizard step. The wizard does not fork on defaults-vs-customize, does not show a provider roster multi-pick, and does not offer a separate implementer-provider pick. Dynamic verifiers are not chosen at launch. `--workers` is compatibility-only, not a launch picker. Repeated `--role-count` / `--role-model` tokens on `renderArgv` are intentional,
202
202
  - **resume-clarification (in-session equivalent)** — there is no separate mode or flag matching the shell's `okstra.sh --resume-clarification`; two steps of the standard flow carry out its substance. (1) `reuse_previous` (yes/no to reuse the previous run's settings — in `requirements-discovery` / `error-analysis` / `implementation-planning`, only when prior run-inputs exist): YES prefills role-count·role-model·directive·related-tasks at once. (2) `clarification_pick`: if the **task-type's own** previous `final-report` exists it is auto-recommended as the carry-in input (falling back to the newest by mtime across all phases when absent), and the same run's `user-responses/` sidecar (answers the user filled in) is attached alongside. The chosen path is passed to prepare as `--clarification-response` — the user makes the sidecar via the report's `Export user response`, places it in `runs/<task-type>/user-responses/`, and re-runs the same phase,
203
203
  - **re-verification scope (`reverify_scope_pick`, `implementation-planning` clarification re-runs only)** — asked right before `confirm` when the re-run is narrowable **or** an answered `C-NNN` traces to no stage. When every answered id traces to a stage: 3 options — `auto` (recommended — leave it to the lead's `okstra incremental-scope` decision) / `full` (re-verify every stage) / Enter directly (a stage-number CSV, validated against the prior report's Stage Map). When an id is unlinked, `auto` is omitted and the user names stages or picks `full`; that unlinked id does not freeze the run at full. The answer goes out as `--reverify-scope` and reaches the lead prompt as the `REVERIFY_SCOPE_MODE` / `REVERIFY_SCOPE_STAGES` tokens; it shapes that CLI's inputs rather than replacing the decision. The confirmation block's `reverify-scope` line names unlinked ids as needing stage numbers, not as a forced full re-run,
204
204
  - `release-handoff` PR template override + persist scope,
@@ -290,6 +290,65 @@ The python function underneath is mutex-protected (`~/.okstra/.locks/<task-key>.
290
290
 
291
291
  You can delete the literal state-file path after this point — its job is done. Invoke `command rm` with the literal path (e.g. `command rm /var/folders/.../okstra-wizard.AbCd.json`), not a shell variable. `command` is what keeps a `rm='rm -i'` alias from turning this into a confirmation prompt nobody is there to answer.
292
292
 
293
+ <!-- BEGIN FRAGMENT: host-orchestration-implementation-planning -->
294
+ ## Host orchestration rules — implementation-planning
295
+
296
+ These are the rules the **host orchestrator** follows around an
297
+ `implementation-planning` run: when a stage genuinely cannot write a RED step,
298
+ and what only the user can grant. They are not lead phase rules — the lead's
299
+ rules live in `prompts/profiles/`.
300
+
301
+ ### Step 5.1 (implementation-planning only): user-confirmed TDD bypass offer
302
+
303
+ Every plan stage must open with a `RED:` step whose outcome is FAIL and reach a
304
+ later `GREEN:` step (validator S10c). `tddExemption` waives that, and until
305
+ 2026-09-10 only for `doc-only`, `config-only`, or `pure-rename` work — so a
306
+ stage that is truthfully none of the three had no passable value, and the plan
307
+ got through by filing the nearest category. That is the failure this flag
308
+ exists to remove: **a closed reason list with no escape makes the plan
309
+ misdescribe itself.**
310
+
311
+ `render-bundle` accepts an optional `--tdd-bypass "<stage>:<reason>"` flag
312
+ (implementation-planning only). It records a **user-acknowledged** bypass into
313
+ `<task-root>/qa/tdd-bypass.json`, and S10e then accepts that stage declaring
314
+ `tddExemption: user-bypass`. The reason is stored **verbatim**.
315
+
316
+ Offer it only when the run's own output says the stage cannot reach RED — a
317
+ planning report blocked at S10e on a `user-bypass` stage, or a plan-body
318
+ verification round whose disagreements say the expected FAIL is unreachable
319
+ (e.g. the stage worktree HEAD is already the accepted commit). Never offer it
320
+ to save a stage that simply has no test written yet; that stage's answer is the
321
+ RED step.
322
+
323
+ This is **never** a lead/worker self-exemption — only the user may grant it,
324
+ and the lead has no command that writes this record. Surface it as a 3-option
325
+ recommendation picker (per the run-prompt recommendation rule):
326
+
327
+ 1. (recommended) Keep the RED/GREEN requirement — re-plan the stage so its
328
+ first step writes the failing test.
329
+ 2. Bypass this stage — ask the user for the stage number and reason, then pass
330
+ `--tdd-bypass "<stage>:<reason>"` to `render-bundle` (reason = the user's
331
+ words, unedited).
332
+ 3. Enter directly — the user types the full `<stage>:<reason>` value.
333
+
334
+ When the user picks a bypass, append `--tdd-bypass "<stage>:<reason>"` to the
335
+ `render-bundle` invocation. Omit the flag entirely otherwise (do **not** pass
336
+ `--tdd-bypass ""`). A malformed value aborts `render-bundle` with a
337
+ `PrepareError`. The grant is per stage number and per task, and it persists
338
+ across runs of that task — the next plan of the same stage may still declare
339
+ `user-bypass` until the user's grant is removed from the ledger.
340
+
341
+ **A stage whose work already landed is usually not a bypass case.** When the
342
+ product change is committed and its conformance result is PASS but the stage
343
+ never registered as done, the honest fix is to close that stage rather than to
344
+ re-plan it without a RED step. Check `okstra stage-map <task-key>` first: when
345
+ `doneStages` omits a stage whose commit is on the stage branch, close it with
346
+ `okstra stage-close <task-key> --stage <N> --from-commit <sha>` and re-plan only
347
+ what is left. That command refuses unless the commit exists and the stage's
348
+ conformance gate permits progress, so it cannot close a stage the run validator
349
+ would have blocked.
350
+ <!-- END FRAGMENT: host-orchestration-implementation-planning -->
351
+
293
352
  <!-- BEGIN FRAGMENT: host-orchestration-implementation -->
294
353
  ## Host orchestration rules — implementation
295
354
 
@@ -31,7 +31,7 @@ generator: okstra-brief-gen
31
31
  <Which tickets are causes, which are observation means, which depend on which. Write `_(none)_` when the briefs' Related Task Graph already says it all.>
32
32
 
33
33
  <!-- okstra:task-memory:begin -->
34
- <!-- okstra rewrites this region at every report-finalize. Edit the sections above it, not this one. -->
34
+ <!-- okstra redraws this region after every report-finalize, at `okstra set-work-status`, and before each run copies this file. Edit the sections above it, not this one. -->
35
35
  ## Task Memory
36
36
 
37
37
  _(no runs recorded yet)_
@@ -21,6 +21,11 @@ for _ssot_dir in (_VALIDATORS_DIR.parent / "scripts", _VALIDATORS_DIR.parent / "
21
21
  sys.path.insert(0, str(_ssot_dir))
22
22
 
23
23
  from okstra_ctl.md_table import split_pipe_row # noqa: E402
24
+ from okstra_ctl.tdd_bypass import ( # noqa: E402
25
+ REASON_TOKEN as TDD_USER_BYPASS_TOKEN,
26
+ bypass_file,
27
+ granted_stages,
28
+ )
24
29
  from okstra_ctl.stage_map import ( # noqa: E402
25
30
  STAGE_MAP_HEADING,
26
31
  StageMapError,
@@ -185,6 +190,15 @@ TDD_EXEMPTION = re.compile(_LABEL_PREFIX + r"TDD exemption\s*:\s*(?:\*\*)?\s*(.+
185
190
  # Profile implementation-planning.md:81 limits the exemption to these three
186
191
  # categories; any other reason (e.g. "refactor") must not waive RED/GREEN.
187
192
  TDD_EXEMPTION_ALLOWED = ("doc-only", "config-only", "pure-rename")
193
+ # The fourth reason is not a category the plan may assert on its own: it holds
194
+ # only while `<task-root>/qa/tdd-bypass.json` records the user's grant for that
195
+ # stage. Every caller therefore passes the granted stage numbers in, and a plan
196
+ # naming the token with no grant fails S10e like any arbitrary reason. Without
197
+ # it a stage that is genuinely none of the three has no passable value and the
198
+ # plan misdescribes itself to get through (2026-09-10, fontsninja-v3-site
199
+ # dev-10628-3: product work already committed and conformance-proved in a prior
200
+ # run of the same stage, filed as `config-only`).
201
+ TDD_EXEMPTION_USER_BYPASS = TDD_USER_BYPASS_TOKEN
188
202
  TEST_CASE_CATEGORIES = ("success", "boundary", "failure")
189
203
  TEST_CASE = {
190
204
  cat: re.compile(
@@ -196,18 +210,48 @@ CONFORMANCE_TESTS = re.compile(_LABEL_PREFIX + r"Conformance tests\s*:\s*(?:\*\*
196
210
  CONFORMANCE_EXEMPTION = re.compile(_LABEL_PREFIX + r"Conformance exemption\s*:\s*(?:\*\*)?\s*\S", re.M)
197
211
 
198
212
 
199
- def _exemption_reason_allowed(section: str) -> bool:
200
- """True when a `TDD exemption:` line is present AND its reason names an
201
- allowed category (doc-only / config-only / pure-rename, case-insensitive).
202
- A present-but-unlisted reason returns False so S10e can reject it."""
213
+ def _exemption_reason_message(stage_number: int) -> str:
214
+ """S10e 의 거절 문구. 통과할 수 있는 값을 전부 이름으로 말한다.
215
+
216
+ 사유 목록만 나열하면 세 카테고리 중 어느 것도 사실이 아닌 stage 는
217
+ 거짓 신고 외에 길이 없다. 네 번째 값과 그 값을 얻는 명령을 함께 적어
218
+ 다음 행동이 문구 안에 있게 한다.
219
+ """
220
+ return (
221
+ "S10e: 'tddExemption' reason must be one of "
222
+ + " / ".join(TDD_EXEMPTION_ALLOWED)
223
+ + f" — or `{TDD_EXEMPTION_USER_BYPASS}`, which holds only while the "
224
+ "user has granted it for this stage: re-run prepare with "
225
+ f'`--tdd-bypass "{stage_number or 1}:<reason>"` so the reason is '
226
+ "recorded verbatim in <task-root>/qa/tdd-bypass.json. An empty or "
227
+ "arbitrary reason cannot waive RED/GREEN and the three test cases"
228
+ )
229
+
230
+
231
+ def _exemption_reason_allowed(
232
+ section: str, *, stage_number: int = 0, user_bypassed: frozenset[int] = frozenset(),
233
+ ) -> bool:
234
+ """True when a `TDD exemption:` line is present AND its reason is allowed.
235
+
236
+ Allowed means one of the three categories (doc-only / config-only /
237
+ pure-rename, case-insensitive), or `user-bypass` for a stage the user
238
+ granted — that grant lives outside the plan, so it is passed in.
239
+ A present-but-unlisted reason returns False so S10e can reject it.
240
+ """
203
241
  m = TDD_EXEMPTION.search(section)
204
242
  if not m:
205
243
  return False
206
244
  reason = m.group(1).lower()
245
+ if TDD_EXEMPTION_USER_BYPASS in reason:
246
+ return stage_number in user_bypassed
207
247
  return any(cat in reason for cat in TDD_EXEMPTION_ALLOWED)
208
248
 
209
249
 
210
- def _check_slice_tdd(text: str, stages: List[StageMapStage]) -> List[ValidationError]:
250
+ def _check_slice_tdd(
251
+ text: str,
252
+ stages: List[StageMapStage],
253
+ user_bypassed: frozenset[int] = frozenset(),
254
+ ) -> List[ValidationError]:
211
255
  """S10: each stage declares a vertical slice and follows RED→GREEN ordering.
212
256
 
213
257
  S10a — `Slice value:` line with a non-empty value.
@@ -237,11 +281,11 @@ def _check_slice_tdd(text: str, stages: List[StageMapStage]) -> List[ValidationE
237
281
  "S10b: 'Acceptance:' line missing or empty"))
238
282
 
239
283
  if TDD_EXEMPTION.search(section):
240
- if not _exemption_reason_allowed(section):
284
+ if not _exemption_reason_allowed(
285
+ section, stage_number=s.stage_number, user_bypassed=user_bypassed,
286
+ ):
241
287
  errs.append(ValidationError("S10", s.stage_number,
242
- "S10e: 'TDD exemption:' reason must be one of "
243
- "doc-only / config-only / pure-rename — an arbitrary reason "
244
- "cannot waive RED/GREEN"))
288
+ _exemption_reason_message(s.stage_number)))
245
289
  continue
246
290
 
247
291
  missing_cases = [
@@ -419,7 +463,9 @@ def _check_parallel_safety(
419
463
  })
420
464
 
421
465
 
422
- def collect_validation_errors(text: str) -> List[ValidationError]:
466
+ def collect_validation_errors(
467
+ text: str, user_bypassed: frozenset[int] = frozenset(),
468
+ ) -> List[ValidationError]:
423
469
  """All S1–S11 checks against the report text; empty list means valid.
424
470
 
425
471
  S1 (missing `## 5.5 Stage Map` heading) makes the rest unparseable, so it
@@ -437,7 +483,7 @@ def collect_validation_errors(text: str) -> List[ValidationError]:
437
483
  return [ValidationError("S2", 0, exc.reason)]
438
484
  if stages:
439
485
  errors.extend(_check_each_stage_section(text, stages))
440
- errors.extend(_check_slice_tdd(text, stages))
486
+ errors.extend(_check_slice_tdd(text, stages, user_bypassed))
441
487
  errors.extend(_check_markdown_step_commands(text, stages))
442
488
  errors.extend(_check_conformance_declaration(text, stages))
443
489
  errors.extend(_check_depends_on(stages))
@@ -465,7 +511,9 @@ def _data_stage_metas(
465
511
  return rows, _stage_numbers_monotonic(rows)
466
512
 
467
513
 
468
- def _check_data_slice_tdd(stage: dict) -> List[ValidationError]:
514
+ def _check_data_slice_tdd(
515
+ stage: dict, user_bypassed: frozenset[int] = frozenset(),
516
+ ) -> List[ValidationError]:
469
517
  """S10c / S10e over one schema-v2 `stages[]` entry.
470
518
 
471
519
  The schema already requires `sliceValue`, `acceptance`, and — through its
@@ -479,12 +527,12 @@ def _check_data_slice_tdd(stage: dict) -> List[ValidationError]:
479
527
  number = stage.get("stage") if isinstance(stage.get("stage"), int) else 0
480
528
  if "tddExemption" in stage:
481
529
  reason = str(stage.get("tddExemption") or "").lower()
530
+ if TDD_EXEMPTION_USER_BYPASS in reason:
531
+ if number in user_bypassed:
532
+ return []
533
+ return [ValidationError("S10", number, _exemption_reason_message(number))]
482
534
  if not any(cat in reason for cat in TDD_EXEMPTION_ALLOWED):
483
- return [ValidationError("S10", number,
484
- "S10e: 'tddExemption' reason must be one of "
485
- + " / ".join(TDD_EXEMPTION_ALLOWED)
486
- + " — an empty or arbitrary reason cannot waive RED/GREEN "
487
- "and the three test cases")]
535
+ return [ValidationError("S10", number, _exemption_reason_message(number))]
488
536
  return []
489
537
 
490
538
  steps = [s for s in (stage.get("stepwiseExecution") or []) if isinstance(s, dict)]
@@ -625,7 +673,9 @@ def _check_data_stage_identities(
625
673
  return errs
626
674
 
627
675
 
628
- def collect_data_validation_errors(planning: dict) -> List[ValidationError]:
676
+ def collect_data_validation_errors(
677
+ planning: dict, user_bypassed: frozenset[int] = frozenset(),
678
+ ) -> List[ValidationError]:
629
679
  """The S-checks that schema v2 cannot express, over `implementationPlanning`.
630
680
 
631
681
  `collect_validation_errors` scans rendered v1 Markdown, which a v2 report
@@ -663,7 +713,7 @@ def collect_data_validation_errors(planning: dict) -> List[ValidationError]:
663
713
  if not meta.depends_on
664
714
  }))
665
715
  for stage in stages:
666
- errors.extend(_check_data_slice_tdd(stage))
716
+ errors.extend(_check_data_slice_tdd(stage, user_bypassed))
667
717
  number = stage.get("stage") if isinstance(stage.get("stage"), int) else 0
668
718
  for step in stage.get("stepwiseExecution") or []:
669
719
  if isinstance(step, dict):
@@ -671,26 +721,62 @@ def collect_data_validation_errors(planning: dict) -> List[ValidationError]:
671
721
  return errors
672
722
 
673
723
 
674
- def collect_plan_errors(plan_path: Path) -> List[ValidationError]:
724
+ def collect_plan_errors(
725
+ plan_path: Path, user_bypassed: frozenset[int] | None = None,
726
+ ) -> List[ValidationError]:
675
727
  """The S-checks for one approved plan, whichever schema wrote it.
676
728
 
677
729
  A schema-v2 report keeps its stage map in the `.data.json` sidecar and
678
730
  renders no `## 5.5 Stage Map` section, so scanning its markdown reports the
679
731
  section as missing and blocks every run that approved such a plan.
680
732
  """
733
+ granted = (
734
+ user_bypassed if user_bypassed is not None
735
+ else user_bypassed_stages_for_plan(plan_path)
736
+ )
681
737
  planning = schema_v2_report(plan_path).get("implementationPlanning")
682
738
  if isinstance(planning, dict) and planning:
683
- return collect_data_validation_errors(planning)
684
- return collect_validation_errors(plan_path.read_text(encoding="utf-8"))
739
+ return collect_data_validation_errors(planning, granted)
740
+ return collect_validation_errors(plan_path.read_text(encoding="utf-8"), granted)
741
+
742
+
743
+ def user_bypassed_stages_for_plan(plan_path: Path) -> frozenset[int]:
744
+ """이 계획이 속한 task 의 우회 원장에서 부여된 stage 번호들.
745
+
746
+ 레이아웃 해석은 `RunRef.from_run_dir` 이 소유한다 — 부모를 세는 계산은
747
+ 배치가 하나만 바뀌어도 조용히 다른 디렉터리를 가리킨다. run 디렉터리로
748
+ 해석되지 않는 입력은 빈 집합이고, 그때는 `--tdd-bypass-ledger` 로 원장을
749
+ 직접 넘기는 것이 경로다. 원장이 없으면 빈 집합 — 우회 없음이다.
750
+ """
751
+ from okstra_ctl.paths import RunRef
752
+
753
+ try:
754
+ task_root = RunRef.from_run_dir(plan_path.resolve().parent.parent).task_root
755
+ except (ValueError, IndexError):
756
+ return frozenset()
757
+ return frozenset(granted_stages(bypass_file(task_root)))
685
758
 
686
759
 
687
760
  def main(argv: List[str]) -> int:
688
761
  p = argparse.ArgumentParser()
689
762
  p.add_argument("--plan", required=True)
763
+ p.add_argument(
764
+ "--tdd-bypass-ledger",
765
+ default="",
766
+ dest="tdd_bypass_ledger",
767
+ help=(
768
+ "path to <task-root>/qa/tdd-bypass.json; defaults to the ledger of "
769
+ "the task the --plan report lives in"
770
+ ),
771
+ )
690
772
  args = p.parse_args(argv)
691
773
 
774
+ granted = (
775
+ frozenset(granted_stages(Path(args.tdd_bypass_ledger)))
776
+ if args.tdd_bypass_ledger else None
777
+ )
692
778
  try:
693
- errors = collect_plan_errors(Path(args.plan))
779
+ errors = collect_plan_errors(Path(args.plan), granted)
694
780
  except StageMapError as exc:
695
781
  print(f"S0 stage=0: {exc.reason}", file=sys.stderr)
696
782
  return 1