okstra 0.196.0 → 0.197.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli-registry.mjs +6 -0
- package/dist/cli-registry.mjs.map +1 -1
- package/docs/cli.md +14 -1
- package/docs/project-structure-overview.md +1 -0
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/host-orchestration/implementation-planning.md +56 -0
- package/runtime/prompts/lead/plan-body-verification.md +21 -3
- package/runtime/prompts/profiles/implementation-planning.md +3 -2
- package/runtime/prompts/wizard/prompts.ko.json +2 -2
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +15 -0
- package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +15 -0
- package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +15 -0
- package/runtime/python/okstra_ctl/agent/prompt_cli/dynamic_verifier.py +12 -10
- package/runtime/python/okstra_ctl/blocking_checks.py +19 -0
- package/runtime/python/okstra_ctl/conformance.py +36 -5
- package/runtime/python/okstra_ctl/dispatch_core.py +11 -6
- package/runtime/python/okstra_ctl/dispatch_state.py +11 -7
- package/runtime/python/okstra_ctl/domain/worker_stream.py +4 -5
- package/runtime/python/okstra_ctl/final_report_schema.py +62 -1
- package/runtime/python/okstra_ctl/group_context.py +96 -3
- package/runtime/python/okstra_ctl/plan_items.py +14 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +45 -3
- package/runtime/python/okstra_ctl/run.py +77 -0
- package/runtime/python/okstra_ctl/session_transcript.py +4 -4
- package/runtime/python/okstra_ctl/set_work_status.py +30 -1
- package/runtime/python/okstra_ctl/stage_close.py +244 -0
- package/runtime/python/okstra_ctl/tdd_bypass.py +131 -0
- package/runtime/python/okstra_ctl/wizard/engine.py +27 -24
- package/runtime/python/okstra_ctl/wizard/picker_navigation.py +37 -14
- package/runtime/python/okstra_ctl/wizard/roles.py +16 -11
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +15 -1
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +45 -2
- package/runtime/skills/okstra-brief-gen/SKILL.md +24 -7
- package/runtime/skills/okstra-inspect/facets/recap.md +17 -1
- package/runtime/skills/okstra-run/SKILL.md +63 -4
- package/runtime/templates/reports/group-context.template.md +1 -1
- package/runtime/validators/validate-implementation-plan-stages.py +109 -23
- package/runtime/validators/validate-run.py +63 -8
|
@@ -83,16 +83,59 @@ WORKER_ERROR_CONTRACT_FILENAME = "worker-error-contract.md"
|
|
|
83
83
|
# 따르되 라운드 원장 밖이라 번호가 없다. 실측(2026-09-09 dev-10642): 이 kind 가
|
|
84
84
|
# 없어 gap 검증 프롬프트를 만들 수 없었고 gap 3건 전부 `gapsUnverified` 로 남았다.
|
|
85
85
|
CRITIC_VERIFY_DISPATCH_KIND = "critic-verify"
|
|
86
|
+
REVERIFY_DISPATCH_KIND_PREFIX = "reverify-r"
|
|
87
|
+
# 계획 본문 검증(§5.5.9) 라운드의 dispatch kind. 결과 파일명 규약이 이미
|
|
88
|
+
# `<role>-worker-plan-verify-r<N>-…` 이라(`plan-body-verification.md` §"Round
|
|
89
|
+
# protocol", `dispatch_state._plan_verify_result_workers`) kind 도 같은 번호를
|
|
90
|
+
# 단다. 실측(2026-09-09 dev-10642 implementation-planning 001): 이 kind 가 없어
|
|
91
|
+
# 계획 항목 큐를 `reverify-r<N>` 으로 보낼 수밖에 없었고, 그러면 검증 계약이
|
|
92
|
+
# `okstra convergence reverify-prompt` 서명을 요구하는데 계획 항목 큐의 정본
|
|
93
|
+
# 렌더러는 `okstra plan-items prompt` 라 통과할 값이 하나도 없었다. 라운드가
|
|
94
|
+
# 한 번도 열리지 못한 채 `planBodyVerification.roundCount` 가 0 으로 남았다.
|
|
95
|
+
PLAN_VERIFY_DISPATCH_KIND_PREFIX = "plan-verify-r"
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _numbered_round(dispatch_kind: str, prefix: str) -> int | None:
|
|
99
|
+
"""`<prefix><N>` 의 N. 접두사가 다르거나 N 이 양의 정수가 아니면 None."""
|
|
100
|
+
if not dispatch_kind.startswith(prefix):
|
|
101
|
+
return None
|
|
102
|
+
suffix = dispatch_kind[len(prefix):]
|
|
103
|
+
if not suffix.isdigit() or int(suffix) < 1:
|
|
104
|
+
return None
|
|
105
|
+
return int(suffix)
|
|
86
106
|
|
|
87
107
|
|
|
88
108
|
def is_verification_dispatch_kind(dispatch_kind: str) -> bool:
|
|
89
|
-
"""번호 reverify 라운드와 critic gap 검증 — 검증 프롬프트 계약을
|
|
109
|
+
"""번호 reverify·계획 본문 라운드와 critic gap 검증 — 검증 프롬프트 계약을
|
|
110
|
+
받는 kind."""
|
|
90
111
|
return (
|
|
91
|
-
dispatch_kind.startswith(
|
|
112
|
+
dispatch_kind.startswith(REVERIFY_DISPATCH_KIND_PREFIX)
|
|
113
|
+
or dispatch_kind.startswith(PLAN_VERIFY_DISPATCH_KIND_PREFIX)
|
|
92
114
|
or dispatch_kind == CRITIC_VERIFY_DISPATCH_KIND
|
|
93
115
|
)
|
|
94
116
|
|
|
95
117
|
|
|
118
|
+
def is_plan_verify_dispatch_kind(dispatch_kind: str) -> bool:
|
|
119
|
+
"""계획 본문 검증 라운드인가. 검증 계약이 요구하는 렌더러 서명이 번호
|
|
120
|
+
reverify 와 다르므로 계약 검사가 이 둘을 갈라야 한다."""
|
|
121
|
+
return _numbered_round(dispatch_kind, PLAN_VERIFY_DISPATCH_KIND_PREFIX) is not None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def verification_dispatch_round(dispatch_kind: str) -> int | None:
|
|
125
|
+
"""검증 kind 가 적는 라운드 번호. 번호 없는 kind(`critic-verify`)는 1,
|
|
126
|
+
검증 kind 가 아니거나 번호가 깨졌으면 None.
|
|
127
|
+
|
|
128
|
+
예약(`agent-prompt materialize`)과 디스패치가 같은 값을 적어야 한다 —
|
|
129
|
+
validate-run 이 team-state 의 kind 와 예약된 invocation 의 `dispatchKind`
|
|
130
|
+
를 대조하므로, 두 계산이 갈리면 그 디스패치가 거부된다."""
|
|
131
|
+
if dispatch_kind == CRITIC_VERIFY_DISPATCH_KIND:
|
|
132
|
+
return 1
|
|
133
|
+
for prefix in (REVERIFY_DISPATCH_KIND_PREFIX, PLAN_VERIFY_DISPATCH_KIND_PREFIX):
|
|
134
|
+
if dispatch_kind.startswith(prefix):
|
|
135
|
+
return _numbered_round(dispatch_kind, prefix)
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
|
|
96
139
|
@dataclass(frozen=True)
|
|
97
140
|
class PromptPlan:
|
|
98
141
|
audience: PromptAudience
|
|
@@ -1024,13 +1024,30 @@ per-page ratio instead of origin load). If the file is absent, ask once via
|
|
|
1024
1024
|
- `Skip — this group needs no shared context`.
|
|
1025
1025
|
|
|
1026
1026
|
Never fill the skeleton yourself from the tickets: a ticket summary decides
|
|
1027
|
-
nothing, and the sections ask for what the tickets do not say.
|
|
1028
|
-
|
|
1029
|
-
file
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
|
|
1033
|
-
|
|
1027
|
+
nothing, and the sections ask for what the tickets do not say.
|
|
1028
|
+
|
|
1029
|
+
If the file exists, what you do with it depends on what you are writing:
|
|
1030
|
+
|
|
1031
|
+
- **A new brief** leaves it untouched. Mention its path in the hand-off block.
|
|
1032
|
+
- **A rewrite or correction of an existing brief** reconciles it in the same
|
|
1033
|
+
response, whenever the change supersedes something the group document
|
|
1034
|
+
states — a Definition of Better figure, a success signal, a Group-Wide
|
|
1035
|
+
Constraint, a Ticket Relation, or what it says is executable from the
|
|
1036
|
+
project root. okstra copies this file into every run's
|
|
1037
|
+
`instruction-set/task-group-context.md` and carries it in the analysis
|
|
1038
|
+
packet **ahead of the brief extract**, so a sentence you corrected in the
|
|
1039
|
+
brief is still outranked by the group document's stale copy of it. Qualify a
|
|
1040
|
+
diverged figure with the date and method of the measurement that contradicts
|
|
1041
|
+
it rather than deleting it, and name the brief the correction came from.
|
|
1042
|
+
|
|
1043
|
+
Either way, edit only the authored sections. The file's trailing `## Task Memory`
|
|
1044
|
+
region (between `<!-- okstra:task-memory:begin -->` and `end`) is okstra's: it
|
|
1045
|
+
holds the group's start order and each task's latest conclusion, and okstra
|
|
1046
|
+
redraws it after every run's `report-finalize`, at `okstra set-work-status`, and
|
|
1047
|
+
before each run copies the file into its instruction set. Anything written there
|
|
1048
|
+
by hand is overwritten without warning. A group whose runs finished before
|
|
1049
|
+
anyone created the file already has one holding that region alone — `init` then
|
|
1050
|
+
inserts the human sections above it.
|
|
1034
1051
|
|
|
1035
1052
|
Then stop. Do not invoke `okstra-run` directly — the user chooses when to
|
|
1036
1053
|
proceed, and they may want to edit the brief externally first. In the
|
|
@@ -6,7 +6,7 @@ Loaded lazily by the dispatch table in `SKILL.md` (core). Shared rules — Step
|
|
|
6
6
|
|
|
7
7
|
Trigger phrases: "okstra recap", "recap", "work summary", "summarize this task", "before/after summary", "explain this work", "task question".
|
|
8
8
|
|
|
9
|
-
On top of the `.okstra` artifacts accumulated for a single task-id — or for every task of one task-group — (a) produce a before/after summary and (b) answer free-form questions about that work. By default it reads only the `.okstra/` subtree (artifact mode). It expands to code mode only when the user explicitly asks to look at the code changes too. This sub-command performs
|
|
9
|
+
On top of the `.okstra` artifacts accumulated for a single task-id — or for every task of one task-group — (a) produce a before/after summary and (b) answer free-form questions about that work. By default it reads only the `.okstra/` subtree (artifact mode). It expands to code mode only when the user explicitly asks to look at the code changes too. This sub-command performs the `recap-log.jsonl` append, the `notes/` note authoring (recap.5), and the group-context reconciliation that note triggers (recap.6); it never mutates `task-manifest.json` / catalog / timeline, and it touches `group-context.md` only in the authored sections above the `<!-- okstra:task-memory:begin -->` marker.
|
|
10
10
|
|
|
11
11
|
### recap.1 — Resolve target
|
|
12
12
|
|
|
@@ -144,3 +144,19 @@ Write the body to the scratchpad as markdown first, then pass it with `--body-fi
|
|
|
144
144
|
4. **`notes/` is inert to okstra** — no run reads it automatically. After writing, relay the `clarificationResponseArg` the CLI printed (e.g. `--clarification-response <notePath>`) to the user verbatim, telling them it only takes effect when the next run is executed with that argument.
|
|
145
145
|
|
|
146
146
|
**Guardrail:** `.okstra/` is gitignored — treat `notes/` as local scratch and never `git add` it. Creating a new note is easy to undo (delete the file), so always prefer it over editing a generated or user-owned file.
|
|
147
|
+
|
|
148
|
+
### recap.6 — Reconcile the task-group context (required whenever recap.5 writes a note)
|
|
149
|
+
|
|
150
|
+
A group's shared context does not stay true while its tasks run. Figures get re-measured, a success signal turns out to be satisfiable without a fix, a constraint names the wrong place. `group-context.md` states in its own header that okstra copies it into each run's `instruction-set/task-group-context.md` and carries it in the analysis packet's `## Task-Group Context` section, **ahead of the brief extract** — so a stale sentence there outranks a corrected brief for every task in the group.
|
|
151
|
+
|
|
152
|
+
**After writing a note, read `<PROJECT_ROOT>/.okstra/briefs/<task-group>/group-context.md` if it exists.** If anything the note establishes touches its Definition of Better figures, success signals, Group-Wide Constraints, Ticket Relations, or its statement of what is executable from the project root, update those sections **in the same response that wrote the note**. A note written without that check is an incomplete deliverable, the same way a `.project-docs/` document without its index row is.
|
|
153
|
+
|
|
154
|
+
Three rules for the edit:
|
|
155
|
+
|
|
156
|
+
1. **Only above the marker.** Edit the authored sections above `<!-- okstra:task-memory:begin -->`. The region below is okstra's projection — start order, per-task status, recorded runs — and it is redrawn from the task manifests at every `report-finalize`, at `okstra set-work-status`, and before each run copies the file. Anything you write there is overwritten without warning.
|
|
157
|
+
2. **Qualify a diverged figure; do not delete it.** An audit number is the baseline for the window it was measured in, not a current reading. Where a direct measurement contradicts it, say so with the measurement's date and method, and keep both. Deleting the old figure destroys the reason the group exists.
|
|
158
|
+
3. **Say which task and note the correction came from.** The next reader needs to get from the sentence back to its evidence.
|
|
159
|
+
|
|
160
|
+
If the group has no `group-context.md`, do not create one here — `okstra group-context init` (via okstra-brief-gen) owns that skeleton, and creating an empty one pre-empts the sections a human meant to author.
|
|
161
|
+
|
|
162
|
+
The same exposure applies to any other edit that supersedes group-wide facts, a brief rewrite most of all. Treat this check as belonging to the fact, not to this sub-command.
|
|
@@ -53,7 +53,7 @@ The wizard tells you which relay operation to use via `next.interaction.kind`. S
|
|
|
53
53
|
- `kind: "done"` → input collection finished; move to Step 5.
|
|
54
54
|
- `kind: "aborted"` → the user picked abort; the wizard is terminally cancelled. Tell the user on one short line that the run setup was aborted, delete the state file (`rm` with the literal path), and stop this skill — do NOT call `render-args` or `render-bundle` (the wizard rejects `render-args` on an aborted state).
|
|
55
55
|
|
|
56
|
-
When native single selection is available, the runtime divides long lists and unsupported multi-selections into selectable screens. Render the returned screen without rebuilding the full list. This applies to plan confirmation, stages, role counts, and provider/model lists. Submit navigation and completion option values normally; the runtime retains the current step until its answer is complete. A ban on textual choice lists does not authorize asking the user to type a model identifier. If a required selector is unavailable, preserve state and follow the relay's recovery. Genuine text steps still collect text.
|
|
56
|
+
When native single selection is available, the runtime divides long lists and unsupported multi-selections into selectable screens. Render the returned screen without rebuilding the full list. This applies to plan confirmation, stages, role counts, and provider/model lists. Submit navigation and completion option values normally; the runtime retains the current step until its answer is complete. A ban on textual choice lists does not authorize asking the user to type a model identifier. If a required selector is unavailable, preserve state and follow the relay's `recovery` object. Genuine text steps still collect text.
|
|
57
57
|
|
|
58
58
|
Submit the answer shape required by `interaction.answerProtocol`; do not add normalization beyond the registered relay's explicit mapping. Invalid, out-of-range, or ambiguous answers return `ok: false` and must re-render the same complete interaction.
|
|
59
59
|
|
|
@@ -83,11 +83,11 @@ On `Okstra preflight: ready`, require `Runtime readiness: ready` before Step 2.
|
|
|
83
83
|
|
|
84
84
|
Carry the fixed `Lead entry mode` line into Step 2's `--entry-mode` as a literal. It is the mode the run must launch in, which is not always the mode this session is in: a host adapter may report that this session cannot host the lead while the run itself is ready. `spawn-process` means okstra starts its own lead process and this session is not the lead — when the line reads `spawn-process` and a readiness check carries `action: spawn-unsandboxed-codex-lead`, tell the user in one line that their current Codex session is sandboxed so okstra will open an unsandboxed lead instead, then continue. Never substitute `current-session` for a `spawn-process` answer, and never recompute the value from the host ID.
|
|
85
85
|
|
|
86
|
-
For a ready response, read the absolute path in the fixed `Relay contract` line with the current host's file-read primitive. Do not derive the path from the host ID or search `PATH`. In that file, find the `Wizard interaction relay` JSON block, require `schemaVersion: 1` and `runtime` equal to the fixed `Runtime` line, then take its `semanticFunctions` allowlist and intersect it with the functions the live harness exposes. The live harness does not expose tools named `native_single_select`. Map each allowlist token to the matching `interactions` kind (`native_single_select` → `native-single`, `native_multi_select` → `native-multi`, `native_question_group` → `native-group`). Include the token in the intersection only when this session can call the string in that kind's `function` field. If the kind is absent from `interactions`, omit the token. Pass only that intersection to Step 2; `plain_text_input` must be present. Keep the parsed `interactions` object for Step 3's function/input/response conversion. An unreadable file, malformed block, runtime mismatch, absent `plain_text_input`, or later interaction kind missing from the object is a host relay contract failure: show the problem and stop rather than guessing.
|
|
86
|
+
For a ready response, read the absolute path in the fixed `Relay contract` line with the current host's file-read primitive. Do not derive the path from the host ID or search `PATH`. In that file, find the `Wizard interaction relay` JSON block, require `schemaVersion: 1` and `runtime` equal to the fixed `Runtime` line, then take its `semanticFunctions` allowlist and intersect it with the functions the live harness exposes. The live harness does not expose tools named `native_single_select`. Map each allowlist token to the matching `interactions` kind (`native_single_select` → `native-single`, `native_multi_select` → `native-multi`, `native_question_group` → `native-group`). Include the token in the intersection only when this session can call the string in that kind's `function` field. If the kind is absent from `interactions`, omit the token. Pass only that intersection to Step 2; `plain_text_input` must be present. Keep the parsed `interactions` object for Step 3's function/input/response conversion, and the parsed `recovery` object for the branch below. An unreadable file, malformed block, runtime mismatch, absent `plain_text_input`, or later interaction kind missing from the object is a host relay contract failure: show the problem and stop rather than guessing.
|
|
87
87
|
|
|
88
88
|
If the successful fixed projection has `Relay contract: -`, enter the compatibility branch below. In that branch only, declare `plain_text_input` and keep its built-in `interactions` mapping for Step 3; do not assume a native tool from the host ID. The existing `unknown command: preflight` branch remains the authoritative stale-CLI failure.
|
|
89
89
|
|
|
90
|
-
Before calculating the effective intersection, apply the registered relay's client-specific tool selection and live mode restrictions. A callable question tool does not necessarily render a selector in the current client. In Codex, use the synchronous picker when permitted; asynchronous question cards are a desktop fallback, not a terminal picker.
|
|
90
|
+
Before calculating the effective intersection, apply the registered relay's client-specific tool selection and live mode restrictions. A callable question tool does not necessarily render a selector in the current client. In Codex, use the synchronous picker when permitted; asynchronous question cards are a desktop fallback, not a terminal picker. A declared native function can still fail at call time: the client refuses the call, the session has no view to present it in, or it returns no answer. That is not a reason to stop, and not a reason to try a different native function. Follow the relay's `recovery.native-question-refused` entry the first time it happens — drop the tokens it names from the intersection, keep `plain_text_input`, tell the user in one line that the host picker is unavailable, and render every remaining screen through the text mapping. Numbered text is that entry's own recovery path, so the ban on printing a numbered list does not apply once it fires. The wizard state file is untouched by a refused call: `okstra wizard step --state-file <path> --no-submit` returns the pending prompt again.
|
|
91
91
|
|
|
92
92
|
Plan adoption (`approve_plan_confirm`) is a workflow choice: present the wizard's existing options using the same selector as other `pick` steps. Follow the relay's distinction between plan decisions and execution permissions; the word "approval" alone is not a reason to replace a selector with a typed confirmation.
|
|
93
93
|
|
|
@@ -198,7 +198,7 @@ That is the entire interactive flow. The wizard handles:
|
|
|
198
198
|
- base-ref pick + git rev-parse validation (skipped when reusing an active worktree),
|
|
199
199
|
- `implementation`-only sub-flow: approved-plan path (frontmatter `approved: true` check) + stage pick (`auto` = the earliest incomplete stage whose dependencies are satisfied, or a specific stage number). Implementer slots use role-count / role-model like every other role (`executor` is only a compatibility alias for `implementer`). When an approved plan is selected and a `## PLAN DECISION` sidecar carrying `Status: approved`, exported from the report — matching the plan on source-report·seq — is detected in that run's sibling `user-responses/`, the approve-confirm step expands to 3 options (`yes_apply` recommended: approve + apply the option as exported / `yes` approve only / `no` abort) — `yes_apply` validates the option against the plan's `optionCandidates` before applying it via the existing approval·option path,
|
|
200
200
|
- `release-handoff`-only sub-flow: after the approved plan auto-resolves, a `handoff_stage_pick` multi-select — choose an eligible stage bundle (stage-group) or the whole task (when an accepted whole-task verification report exists); the result goes out as render-args' `stages` key (csv, empty when whole-task),
|
|
201
|
-
- launch selection after identity/worktree steps: one screen per static role, in profile order. A role that can run several instances (`max > 1`) is a checkbox step `role-models:<role>` (`multi: true`) — the number of models checked is the number of instances, there is no separate count question; the label states the profile range and recommended count, every executable candidate is listed on that one screen (defaults first, the recommended set flagged), and when the list exceeds the host's native checkbox limit the runtime either splits it into several checkbox questions on one `pick_group` screen (interaction plan `native-group`; the question steps are `role-models:<role>#1`, `#2`, … and the answer is one JSON object keyed by them, each value a CSV) when the host's native question group holds every option, or returns the interaction plan `numbered-multi` — render the whole list, never a shortlist or pages. An optional role (`min = 0`, e.g. critic) carries a `추가 안 함` row. A fixed single role (`min = max = 1`, e.g. report-writer) is a single pick `role-model:<role>:1
|
|
201
|
+
- launch selection after identity/worktree steps: one screen per static role, in profile order. A role that can run several instances (`max > 1`) is a checkbox step `role-models:<role>` (`multi: true`) — the number of models checked is the number of instances, there is no separate count question; the label states the profile range and recommended count, every executable candidate is listed on that one screen (defaults first, the recommended set flagged), and when the list exceeds the host's native checkbox limit the runtime either splits it into several checkbox questions on one `pick_group` screen (interaction plan `native-group`; the question steps are `role-models:<role>#1`, `#2`, … and the answer is one JSON object keyed by them, each value a CSV) when the host's native question group holds every option, or returns the interaction plan `numbered-multi` — render the whole list, never a shortlist or pages. An optional role (`min = 0`, e.g. critic) carries a `추가 안 함` row. A fixed single role (`min = max = 1`, e.g. report-writer) is a single pick `role-model:<role>:1`, and a role capped at one model (`max = 1`, e.g. critic) is a single pick on its `role-models:<role>` step; both split the same way when they exceed the native option limit — the tabs are checkbox questions, the answer is the same keyed JSON object, and the wizard rejects a screen whose tabs together name more than one value. current-session lead is this session and is listed on the confirmation summary, not as a wizard step. The wizard does not fork on defaults-vs-customize, does not show a provider roster multi-pick, and does not offer a separate implementer-provider pick. Dynamic verifiers are not chosen at launch. `--workers` is compatibility-only, not a launch picker. Repeated `--role-count` / `--role-model` tokens on `renderArgv` are intentional,
|
|
202
202
|
- **resume-clarification (in-session equivalent)** — there is no separate mode or flag matching the shell's `okstra.sh --resume-clarification`; two steps of the standard flow carry out its substance. (1) `reuse_previous` (yes/no to reuse the previous run's settings — in `requirements-discovery` / `error-analysis` / `implementation-planning`, only when prior run-inputs exist): YES prefills role-count·role-model·directive·related-tasks at once. (2) `clarification_pick`: if the **task-type's own** previous `final-report` exists it is auto-recommended as the carry-in input (falling back to the newest by mtime across all phases when absent), and the same run's `user-responses/` sidecar (answers the user filled in) is attached alongside. The chosen path is passed to prepare as `--clarification-response` — the user makes the sidecar via the report's `Export user response`, places it in `runs/<task-type>/user-responses/`, and re-runs the same phase,
|
|
203
203
|
- **re-verification scope (`reverify_scope_pick`, `implementation-planning` clarification re-runs only)** — asked right before `confirm` when the re-run is narrowable **or** an answered `C-NNN` traces to no stage. When every answered id traces to a stage: 3 options — `auto` (recommended — leave it to the lead's `okstra incremental-scope` decision) / `full` (re-verify every stage) / Enter directly (a stage-number CSV, validated against the prior report's Stage Map). When an id is unlinked, `auto` is omitted and the user names stages or picks `full`; that unlinked id does not freeze the run at full. The answer goes out as `--reverify-scope` and reaches the lead prompt as the `REVERIFY_SCOPE_MODE` / `REVERIFY_SCOPE_STAGES` tokens; it shapes that CLI's inputs rather than replacing the decision. The confirmation block's `reverify-scope` line names unlinked ids as needing stage numbers, not as a forced full re-run,
|
|
204
204
|
- `release-handoff` PR template override + persist scope,
|
|
@@ -290,6 +290,65 @@ The python function underneath is mutex-protected (`~/.okstra/.locks/<task-key>.
|
|
|
290
290
|
|
|
291
291
|
You can delete the literal state-file path after this point — its job is done. Invoke `command rm` with the literal path (e.g. `command rm /var/folders/.../okstra-wizard.AbCd.json`), not a shell variable. `command` is what keeps a `rm='rm -i'` alias from turning this into a confirmation prompt nobody is there to answer.
|
|
292
292
|
|
|
293
|
+
<!-- BEGIN FRAGMENT: host-orchestration-implementation-planning -->
|
|
294
|
+
## Host orchestration rules — implementation-planning
|
|
295
|
+
|
|
296
|
+
These are the rules the **host orchestrator** follows around an
|
|
297
|
+
`implementation-planning` run: when a stage genuinely cannot write a RED step,
|
|
298
|
+
and what only the user can grant. They are not lead phase rules — the lead's
|
|
299
|
+
rules live in `prompts/profiles/`.
|
|
300
|
+
|
|
301
|
+
### Step 5.1 (implementation-planning only): user-confirmed TDD bypass offer
|
|
302
|
+
|
|
303
|
+
Every plan stage must open with a `RED:` step whose outcome is FAIL and reach a
|
|
304
|
+
later `GREEN:` step (validator S10c). `tddExemption` waives that, and until
|
|
305
|
+
2026-09-10 only for `doc-only`, `config-only`, or `pure-rename` work — so a
|
|
306
|
+
stage that is truthfully none of the three had no passable value, and the plan
|
|
307
|
+
got through by filing the nearest category. That is the failure this flag
|
|
308
|
+
exists to remove: **a closed reason list with no escape makes the plan
|
|
309
|
+
misdescribe itself.**
|
|
310
|
+
|
|
311
|
+
`render-bundle` accepts an optional `--tdd-bypass "<stage>:<reason>"` flag
|
|
312
|
+
(implementation-planning only). It records a **user-acknowledged** bypass into
|
|
313
|
+
`<task-root>/qa/tdd-bypass.json`, and S10e then accepts that stage declaring
|
|
314
|
+
`tddExemption: user-bypass`. The reason is stored **verbatim**.
|
|
315
|
+
|
|
316
|
+
Offer it only when the run's own output says the stage cannot reach RED — a
|
|
317
|
+
planning report blocked at S10e on a `user-bypass` stage, or a plan-body
|
|
318
|
+
verification round whose disagreements say the expected FAIL is unreachable
|
|
319
|
+
(e.g. the stage worktree HEAD is already the accepted commit). Never offer it
|
|
320
|
+
to save a stage that simply has no test written yet; that stage's answer is the
|
|
321
|
+
RED step.
|
|
322
|
+
|
|
323
|
+
This is **never** a lead/worker self-exemption — only the user may grant it,
|
|
324
|
+
and the lead has no command that writes this record. Surface it as a 3-option
|
|
325
|
+
recommendation picker (per the run-prompt recommendation rule):
|
|
326
|
+
|
|
327
|
+
1. (recommended) Keep the RED/GREEN requirement — re-plan the stage so its
|
|
328
|
+
first step writes the failing test.
|
|
329
|
+
2. Bypass this stage — ask the user for the stage number and reason, then pass
|
|
330
|
+
`--tdd-bypass "<stage>:<reason>"` to `render-bundle` (reason = the user's
|
|
331
|
+
words, unedited).
|
|
332
|
+
3. Enter directly — the user types the full `<stage>:<reason>` value.
|
|
333
|
+
|
|
334
|
+
When the user picks a bypass, append `--tdd-bypass "<stage>:<reason>"` to the
|
|
335
|
+
`render-bundle` invocation. Omit the flag entirely otherwise (do **not** pass
|
|
336
|
+
`--tdd-bypass ""`). A malformed value aborts `render-bundle` with a
|
|
337
|
+
`PrepareError`. The grant is per stage number and per task, and it persists
|
|
338
|
+
across runs of that task — the next plan of the same stage may still declare
|
|
339
|
+
`user-bypass` until the user's grant is removed from the ledger.
|
|
340
|
+
|
|
341
|
+
**A stage whose work already landed is usually not a bypass case.** When the
|
|
342
|
+
product change is committed and its conformance result is PASS but the stage
|
|
343
|
+
never registered as done, the honest fix is to close that stage rather than to
|
|
344
|
+
re-plan it without a RED step. Check `okstra stage-map <task-key>` first: when
|
|
345
|
+
`doneStages` omits a stage whose commit is on the stage branch, close it with
|
|
346
|
+
`okstra stage-close <task-key> --stage <N> --from-commit <sha>` and re-plan only
|
|
347
|
+
what is left. That command refuses unless the commit exists and the stage's
|
|
348
|
+
conformance gate permits progress, so it cannot close a stage the run validator
|
|
349
|
+
would have blocked.
|
|
350
|
+
<!-- END FRAGMENT: host-orchestration-implementation-planning -->
|
|
351
|
+
|
|
293
352
|
<!-- BEGIN FRAGMENT: host-orchestration-implementation -->
|
|
294
353
|
## Host orchestration rules — implementation
|
|
295
354
|
|
|
@@ -31,7 +31,7 @@ generator: okstra-brief-gen
|
|
|
31
31
|
<Which tickets are causes, which are observation means, which depend on which. Write `_(none)_` when the briefs' Related Task Graph already says it all.>
|
|
32
32
|
|
|
33
33
|
<!-- okstra:task-memory:begin -->
|
|
34
|
-
<!-- okstra
|
|
34
|
+
<!-- okstra redraws this region after every report-finalize, at `okstra set-work-status`, and before each run copies this file. Edit the sections above it, not this one. -->
|
|
35
35
|
## Task Memory
|
|
36
36
|
|
|
37
37
|
_(no runs recorded yet)_
|
|
@@ -21,6 +21,11 @@ for _ssot_dir in (_VALIDATORS_DIR.parent / "scripts", _VALIDATORS_DIR.parent / "
|
|
|
21
21
|
sys.path.insert(0, str(_ssot_dir))
|
|
22
22
|
|
|
23
23
|
from okstra_ctl.md_table import split_pipe_row # noqa: E402
|
|
24
|
+
from okstra_ctl.tdd_bypass import ( # noqa: E402
|
|
25
|
+
REASON_TOKEN as TDD_USER_BYPASS_TOKEN,
|
|
26
|
+
bypass_file,
|
|
27
|
+
granted_stages,
|
|
28
|
+
)
|
|
24
29
|
from okstra_ctl.stage_map import ( # noqa: E402
|
|
25
30
|
STAGE_MAP_HEADING,
|
|
26
31
|
StageMapError,
|
|
@@ -185,6 +190,15 @@ TDD_EXEMPTION = re.compile(_LABEL_PREFIX + r"TDD exemption\s*:\s*(?:\*\*)?\s*(.+
|
|
|
185
190
|
# Profile implementation-planning.md:81 limits the exemption to these three
|
|
186
191
|
# categories; any other reason (e.g. "refactor") must not waive RED/GREEN.
|
|
187
192
|
TDD_EXEMPTION_ALLOWED = ("doc-only", "config-only", "pure-rename")
|
|
193
|
+
# The fourth reason is not a category the plan may assert on its own: it holds
|
|
194
|
+
# only while `<task-root>/qa/tdd-bypass.json` records the user's grant for that
|
|
195
|
+
# stage. Every caller therefore passes the granted stage numbers in, and a plan
|
|
196
|
+
# naming the token with no grant fails S10e like any arbitrary reason. Without
|
|
197
|
+
# it a stage that is genuinely none of the three has no passable value and the
|
|
198
|
+
# plan misdescribes itself to get through (2026-09-10, fontsninja-v3-site
|
|
199
|
+
# dev-10628-3: product work already committed and conformance-proved in a prior
|
|
200
|
+
# run of the same stage, filed as `config-only`).
|
|
201
|
+
TDD_EXEMPTION_USER_BYPASS = TDD_USER_BYPASS_TOKEN
|
|
188
202
|
TEST_CASE_CATEGORIES = ("success", "boundary", "failure")
|
|
189
203
|
TEST_CASE = {
|
|
190
204
|
cat: re.compile(
|
|
@@ -196,18 +210,48 @@ CONFORMANCE_TESTS = re.compile(_LABEL_PREFIX + r"Conformance tests\s*:\s*(?:\*\*
|
|
|
196
210
|
CONFORMANCE_EXEMPTION = re.compile(_LABEL_PREFIX + r"Conformance exemption\s*:\s*(?:\*\*)?\s*\S", re.M)
|
|
197
211
|
|
|
198
212
|
|
|
199
|
-
def
|
|
200
|
-
"""
|
|
201
|
-
|
|
202
|
-
|
|
213
|
+
def _exemption_reason_message(stage_number: int) -> str:
|
|
214
|
+
"""S10e 의 거절 문구. 통과할 수 있는 값을 전부 이름으로 말한다.
|
|
215
|
+
|
|
216
|
+
사유 목록만 나열하면 세 카테고리 중 어느 것도 사실이 아닌 stage 는
|
|
217
|
+
거짓 신고 외에 길이 없다. 네 번째 값과 그 값을 얻는 명령을 함께 적어
|
|
218
|
+
다음 행동이 문구 안에 있게 한다.
|
|
219
|
+
"""
|
|
220
|
+
return (
|
|
221
|
+
"S10e: 'tddExemption' reason must be one of "
|
|
222
|
+
+ " / ".join(TDD_EXEMPTION_ALLOWED)
|
|
223
|
+
+ f" — or `{TDD_EXEMPTION_USER_BYPASS}`, which holds only while the "
|
|
224
|
+
"user has granted it for this stage: re-run prepare with "
|
|
225
|
+
f'`--tdd-bypass "{stage_number or 1}:<reason>"` so the reason is '
|
|
226
|
+
"recorded verbatim in <task-root>/qa/tdd-bypass.json. An empty or "
|
|
227
|
+
"arbitrary reason cannot waive RED/GREEN and the three test cases"
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _exemption_reason_allowed(
|
|
232
|
+
section: str, *, stage_number: int = 0, user_bypassed: frozenset[int] = frozenset(),
|
|
233
|
+
) -> bool:
|
|
234
|
+
"""True when a `TDD exemption:` line is present AND its reason is allowed.
|
|
235
|
+
|
|
236
|
+
Allowed means one of the three categories (doc-only / config-only /
|
|
237
|
+
pure-rename, case-insensitive), or `user-bypass` for a stage the user
|
|
238
|
+
granted — that grant lives outside the plan, so it is passed in.
|
|
239
|
+
A present-but-unlisted reason returns False so S10e can reject it.
|
|
240
|
+
"""
|
|
203
241
|
m = TDD_EXEMPTION.search(section)
|
|
204
242
|
if not m:
|
|
205
243
|
return False
|
|
206
244
|
reason = m.group(1).lower()
|
|
245
|
+
if TDD_EXEMPTION_USER_BYPASS in reason:
|
|
246
|
+
return stage_number in user_bypassed
|
|
207
247
|
return any(cat in reason for cat in TDD_EXEMPTION_ALLOWED)
|
|
208
248
|
|
|
209
249
|
|
|
210
|
-
def _check_slice_tdd(
|
|
250
|
+
def _check_slice_tdd(
|
|
251
|
+
text: str,
|
|
252
|
+
stages: List[StageMapStage],
|
|
253
|
+
user_bypassed: frozenset[int] = frozenset(),
|
|
254
|
+
) -> List[ValidationError]:
|
|
211
255
|
"""S10: each stage declares a vertical slice and follows RED→GREEN ordering.
|
|
212
256
|
|
|
213
257
|
S10a — `Slice value:` line with a non-empty value.
|
|
@@ -237,11 +281,11 @@ def _check_slice_tdd(text: str, stages: List[StageMapStage]) -> List[ValidationE
|
|
|
237
281
|
"S10b: 'Acceptance:' line missing or empty"))
|
|
238
282
|
|
|
239
283
|
if TDD_EXEMPTION.search(section):
|
|
240
|
-
if not _exemption_reason_allowed(
|
|
284
|
+
if not _exemption_reason_allowed(
|
|
285
|
+
section, stage_number=s.stage_number, user_bypassed=user_bypassed,
|
|
286
|
+
):
|
|
241
287
|
errs.append(ValidationError("S10", s.stage_number,
|
|
242
|
-
|
|
243
|
-
"doc-only / config-only / pure-rename — an arbitrary reason "
|
|
244
|
-
"cannot waive RED/GREEN"))
|
|
288
|
+
_exemption_reason_message(s.stage_number)))
|
|
245
289
|
continue
|
|
246
290
|
|
|
247
291
|
missing_cases = [
|
|
@@ -419,7 +463,9 @@ def _check_parallel_safety(
|
|
|
419
463
|
})
|
|
420
464
|
|
|
421
465
|
|
|
422
|
-
def collect_validation_errors(
|
|
466
|
+
def collect_validation_errors(
|
|
467
|
+
text: str, user_bypassed: frozenset[int] = frozenset(),
|
|
468
|
+
) -> List[ValidationError]:
|
|
423
469
|
"""All S1–S11 checks against the report text; empty list means valid.
|
|
424
470
|
|
|
425
471
|
S1 (missing `## 5.5 Stage Map` heading) makes the rest unparseable, so it
|
|
@@ -437,7 +483,7 @@ def collect_validation_errors(text: str) -> List[ValidationError]:
|
|
|
437
483
|
return [ValidationError("S2", 0, exc.reason)]
|
|
438
484
|
if stages:
|
|
439
485
|
errors.extend(_check_each_stage_section(text, stages))
|
|
440
|
-
errors.extend(_check_slice_tdd(text, stages))
|
|
486
|
+
errors.extend(_check_slice_tdd(text, stages, user_bypassed))
|
|
441
487
|
errors.extend(_check_markdown_step_commands(text, stages))
|
|
442
488
|
errors.extend(_check_conformance_declaration(text, stages))
|
|
443
489
|
errors.extend(_check_depends_on(stages))
|
|
@@ -465,7 +511,9 @@ def _data_stage_metas(
|
|
|
465
511
|
return rows, _stage_numbers_monotonic(rows)
|
|
466
512
|
|
|
467
513
|
|
|
468
|
-
def _check_data_slice_tdd(
|
|
514
|
+
def _check_data_slice_tdd(
|
|
515
|
+
stage: dict, user_bypassed: frozenset[int] = frozenset(),
|
|
516
|
+
) -> List[ValidationError]:
|
|
469
517
|
"""S10c / S10e over one schema-v2 `stages[]` entry.
|
|
470
518
|
|
|
471
519
|
The schema already requires `sliceValue`, `acceptance`, and — through its
|
|
@@ -479,12 +527,12 @@ def _check_data_slice_tdd(stage: dict) -> List[ValidationError]:
|
|
|
479
527
|
number = stage.get("stage") if isinstance(stage.get("stage"), int) else 0
|
|
480
528
|
if "tddExemption" in stage:
|
|
481
529
|
reason = str(stage.get("tddExemption") or "").lower()
|
|
530
|
+
if TDD_EXEMPTION_USER_BYPASS in reason:
|
|
531
|
+
if number in user_bypassed:
|
|
532
|
+
return []
|
|
533
|
+
return [ValidationError("S10", number, _exemption_reason_message(number))]
|
|
482
534
|
if not any(cat in reason for cat in TDD_EXEMPTION_ALLOWED):
|
|
483
|
-
return [ValidationError("S10", number,
|
|
484
|
-
"S10e: 'tddExemption' reason must be one of "
|
|
485
|
-
+ " / ".join(TDD_EXEMPTION_ALLOWED)
|
|
486
|
-
+ " — an empty or arbitrary reason cannot waive RED/GREEN "
|
|
487
|
-
"and the three test cases")]
|
|
535
|
+
return [ValidationError("S10", number, _exemption_reason_message(number))]
|
|
488
536
|
return []
|
|
489
537
|
|
|
490
538
|
steps = [s for s in (stage.get("stepwiseExecution") or []) if isinstance(s, dict)]
|
|
@@ -625,7 +673,9 @@ def _check_data_stage_identities(
|
|
|
625
673
|
return errs
|
|
626
674
|
|
|
627
675
|
|
|
628
|
-
def collect_data_validation_errors(
|
|
676
|
+
def collect_data_validation_errors(
|
|
677
|
+
planning: dict, user_bypassed: frozenset[int] = frozenset(),
|
|
678
|
+
) -> List[ValidationError]:
|
|
629
679
|
"""The S-checks that schema v2 cannot express, over `implementationPlanning`.
|
|
630
680
|
|
|
631
681
|
`collect_validation_errors` scans rendered v1 Markdown, which a v2 report
|
|
@@ -663,7 +713,7 @@ def collect_data_validation_errors(planning: dict) -> List[ValidationError]:
|
|
|
663
713
|
if not meta.depends_on
|
|
664
714
|
}))
|
|
665
715
|
for stage in stages:
|
|
666
|
-
errors.extend(_check_data_slice_tdd(stage))
|
|
716
|
+
errors.extend(_check_data_slice_tdd(stage, user_bypassed))
|
|
667
717
|
number = stage.get("stage") if isinstance(stage.get("stage"), int) else 0
|
|
668
718
|
for step in stage.get("stepwiseExecution") or []:
|
|
669
719
|
if isinstance(step, dict):
|
|
@@ -671,26 +721,62 @@ def collect_data_validation_errors(planning: dict) -> List[ValidationError]:
|
|
|
671
721
|
return errors
|
|
672
722
|
|
|
673
723
|
|
|
674
|
-
def collect_plan_errors(
|
|
724
|
+
def collect_plan_errors(
|
|
725
|
+
plan_path: Path, user_bypassed: frozenset[int] | None = None,
|
|
726
|
+
) -> List[ValidationError]:
|
|
675
727
|
"""The S-checks for one approved plan, whichever schema wrote it.
|
|
676
728
|
|
|
677
729
|
A schema-v2 report keeps its stage map in the `.data.json` sidecar and
|
|
678
730
|
renders no `## 5.5 Stage Map` section, so scanning its markdown reports the
|
|
679
731
|
section as missing and blocks every run that approved such a plan.
|
|
680
732
|
"""
|
|
733
|
+
granted = (
|
|
734
|
+
user_bypassed if user_bypassed is not None
|
|
735
|
+
else user_bypassed_stages_for_plan(plan_path)
|
|
736
|
+
)
|
|
681
737
|
planning = schema_v2_report(plan_path).get("implementationPlanning")
|
|
682
738
|
if isinstance(planning, dict) and planning:
|
|
683
|
-
return collect_data_validation_errors(planning)
|
|
684
|
-
return collect_validation_errors(plan_path.read_text(encoding="utf-8"))
|
|
739
|
+
return collect_data_validation_errors(planning, granted)
|
|
740
|
+
return collect_validation_errors(plan_path.read_text(encoding="utf-8"), granted)
|
|
741
|
+
|
|
742
|
+
|
|
743
|
+
def user_bypassed_stages_for_plan(plan_path: Path) -> frozenset[int]:
|
|
744
|
+
"""이 계획이 속한 task 의 우회 원장에서 부여된 stage 번호들.
|
|
745
|
+
|
|
746
|
+
레이아웃 해석은 `RunRef.from_run_dir` 이 소유한다 — 부모를 세는 계산은
|
|
747
|
+
배치가 하나만 바뀌어도 조용히 다른 디렉터리를 가리킨다. run 디렉터리로
|
|
748
|
+
해석되지 않는 입력은 빈 집합이고, 그때는 `--tdd-bypass-ledger` 로 원장을
|
|
749
|
+
직접 넘기는 것이 경로다. 원장이 없으면 빈 집합 — 우회 없음이다.
|
|
750
|
+
"""
|
|
751
|
+
from okstra_ctl.paths import RunRef
|
|
752
|
+
|
|
753
|
+
try:
|
|
754
|
+
task_root = RunRef.from_run_dir(plan_path.resolve().parent.parent).task_root
|
|
755
|
+
except (ValueError, IndexError):
|
|
756
|
+
return frozenset()
|
|
757
|
+
return frozenset(granted_stages(bypass_file(task_root)))
|
|
685
758
|
|
|
686
759
|
|
|
687
760
|
def main(argv: List[str]) -> int:
|
|
688
761
|
p = argparse.ArgumentParser()
|
|
689
762
|
p.add_argument("--plan", required=True)
|
|
763
|
+
p.add_argument(
|
|
764
|
+
"--tdd-bypass-ledger",
|
|
765
|
+
default="",
|
|
766
|
+
dest="tdd_bypass_ledger",
|
|
767
|
+
help=(
|
|
768
|
+
"path to <task-root>/qa/tdd-bypass.json; defaults to the ledger of "
|
|
769
|
+
"the task the --plan report lives in"
|
|
770
|
+
),
|
|
771
|
+
)
|
|
690
772
|
args = p.parse_args(argv)
|
|
691
773
|
|
|
774
|
+
granted = (
|
|
775
|
+
frozenset(granted_stages(Path(args.tdd_bypass_ledger)))
|
|
776
|
+
if args.tdd_bypass_ledger else None
|
|
777
|
+
)
|
|
692
778
|
try:
|
|
693
|
-
errors = collect_plan_errors(Path(args.plan))
|
|
779
|
+
errors = collect_plan_errors(Path(args.plan), granted)
|
|
694
780
|
except StageMapError as exc:
|
|
695
781
|
print(f"S0 stage=0: {exc.reason}", file=sys.stderr)
|
|
696
782
|
return 1
|