okstra 0.191.2 → 0.193.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +2 -2
- package/docs/cli.md +3 -2
- package/docs/project-structure-overview.md +3 -1
- package/docs/task-process/README.md +2 -2
- package/docs/task-process/common-flow.md +4 -5
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/translator-worker.md +1 -1
- package/runtime/prompts/launch.template.md +1 -1
- package/runtime/prompts/lead/convergence.md +13 -3
- package/runtime/prompts/lead/okstra-lead-contract.md +2 -2
- package/runtime/prompts/lead/plan-body-verification.md +1 -1
- package/runtime/prompts/lead/report-writer.md +11 -8
- package/runtime/prompts/profiles/_implementation-executor.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +52 -23
- package/runtime/python/okstra_ctl/conformance.py +74 -0
- package/runtime/python/okstra_ctl/convergence.py +63 -1
- package/runtime/python/okstra_ctl/convergence_reverify_prompt.py +238 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +42 -22
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +96 -8
- package/runtime/python/okstra_ctl/next_phase.py +18 -8
- package/runtime/python/okstra_ctl/plan_items.py +6 -4
- package/runtime/python/okstra_ctl/plan_items_cli.py +91 -6
- package/runtime/python/okstra_ctl/report_finalize.py +57 -10
- package/runtime/python/okstra_ctl/report_translation_dispatch.py +300 -0
- package/runtime/python/okstra_ctl/verdict_blocks.py +37 -7
- package/runtime/python/okstra_ctl/wizard/engine.py +16 -2
- package/runtime/python/okstra_ctl/wizard/registry.py +11 -2
- package/runtime/python/okstra_ctl/wizard/roles.py +364 -361
- package/runtime/python/okstra_ctl/wizard/state.py +39 -27
- package/runtime/python/okstra_ctl/wizard/steps_identity.py +50 -8
- package/runtime/python/okstra_ctl/wizard/steps_roles.py +1 -0
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +11 -0
- package/runtime/skills/okstra-run/SKILL.md +2 -2
- package/runtime/validators/validate-run.py +78 -16
|
@@ -34,6 +34,8 @@ class WizardState:
|
|
|
34
34
|
role_counts: dict[str, int] = field(default_factory=dict)
|
|
35
35
|
role_models: dict[str, list[str]] = field(default_factory=dict)
|
|
36
36
|
role_selection_order: list[str] = field(default_factory=list)
|
|
37
|
+
# 체크박스 화면에서 "직접 선택" 을 골라 전체 후보 화면을 기다리는 역할들.
|
|
38
|
+
role_models_custom: list[str] = field(default_factory=list)
|
|
37
39
|
|
|
38
40
|
# bootstrap
|
|
39
41
|
workspace_root: str = ""
|
|
@@ -212,32 +214,43 @@ class Prompt:
|
|
|
212
214
|
막는다.
|
|
213
215
|
"""
|
|
214
216
|
values = [option.value for option in self.options]
|
|
215
|
-
if PICK_TYPE_CUSTOM
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
217
|
+
if PICK_TYPE_CUSTOM in values:
|
|
218
|
+
after = values[values.index(PICK_TYPE_CUSTOM) + 1:]
|
|
219
|
+
offenders = [value for value in after if value != _ABORT_OPTION]
|
|
220
|
+
if offenders:
|
|
221
|
+
raise WizardError(
|
|
222
|
+
f"wizard step {self.step!r}: the free-input option "
|
|
223
|
+
f"({PICK_TYPE_CUSTOM!r}) must come after every real choice — "
|
|
224
|
+
f"only {_ABORT_OPTION!r} may follow it, but {offenders} do"
|
|
225
|
+
)
|
|
226
|
+
# 추천 불변식은 탈출구가 없는 목록에도 적용된다.
|
|
225
227
|
self._check_recommendations()
|
|
226
228
|
|
|
227
229
|
def _check_recommendations(self) -> None:
|
|
228
|
-
"""추천은
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
230
|
+
"""단일 선택의 추천은 정확히 하나이고 1번이다. 탈출구는 추천이 아니다.
|
|
231
|
+
|
|
232
|
+
실측(2026-09-09, task 선택 화면): 남은 task 세 줄이 전부 `(추천)` 을
|
|
233
|
+
달고 나왔고, 리드는 산문에서 2번을 권했다. 추천이 여럿이면 라벨은
|
|
234
|
+
아무것도 고르지 않은 것이고, 추천이 1번이 아니면 사용자는 목록을
|
|
235
|
+
끝까지 읽어야 추천을 찾는다. 체크박스(`multi`)는 추천이 기본 선택
|
|
236
|
+
집합이라 여럿일 수 있되 앞머리에 모여 있다. 그리고 `직접 입력` /
|
|
237
|
+
`중단` 은 앞의 선택지가 전부 맞지 않을 때의 탈출구이므로 추천 대상이
|
|
238
|
+
될 수 없다.
|
|
233
239
|
"""
|
|
234
240
|
flags = [option.recommended for option in self.options]
|
|
235
|
-
if any(flags)
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
+
if any(flags):
|
|
242
|
+
if self.multi:
|
|
243
|
+
if any(flags[index] for index in range(1, len(flags))
|
|
244
|
+
if not flags[index - 1]):
|
|
245
|
+
raise WizardError(
|
|
246
|
+
f"wizard step {self.step!r}: recommended options must "
|
|
247
|
+
"be the leading run of the list"
|
|
248
|
+
)
|
|
249
|
+
elif sum(flags) != 1 or not flags[0]:
|
|
250
|
+
raise WizardError(
|
|
251
|
+
f"wizard step {self.step!r}: a single-select step carries "
|
|
252
|
+
"exactly one recommendation and it is the first option"
|
|
253
|
+
)
|
|
241
254
|
escapes = {PICK_TYPE_CUSTOM, _ABORT_OPTION}
|
|
242
255
|
marked = [
|
|
243
256
|
option.value for option in self.options
|
|
@@ -436,8 +449,6 @@ def _convert_v1_provider_selections(
|
|
|
436
449
|
)
|
|
437
450
|
if cross_requirement.min_count < cross_requirement.max_count:
|
|
438
451
|
state.role_counts[cross_requirement.role] = selected_count
|
|
439
|
-
count_id = f"role-count:{cross_requirement.role}"
|
|
440
|
-
state.role_selection_order.append(count_id)
|
|
441
452
|
if not providers:
|
|
442
453
|
providers = [
|
|
443
454
|
model.provider_id
|
|
@@ -458,9 +469,9 @@ def _convert_v1_provider_selections(
|
|
|
458
469
|
if model_ref is None:
|
|
459
470
|
break
|
|
460
471
|
state.role_models.setdefault(cross_requirement.role, []).append(model_ref)
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
)
|
|
472
|
+
if cross_requirement.role in state.role_models:
|
|
473
|
+
# 여러 인스턴스 역할의 답은 체크박스 한 장이다.
|
|
474
|
+
state.role_selection_order.append(f"role-models:{cross_requirement.role}")
|
|
464
475
|
|
|
465
476
|
provider_rows = (
|
|
466
477
|
(
|
|
@@ -572,7 +583,7 @@ def _validate_v2_state_fields(data: dict[str, Any]) -> None:
|
|
|
572
583
|
|
|
573
584
|
|
|
574
585
|
def _is_role_selection_step(step_id: str) -> bool:
|
|
575
|
-
return step_id.startswith(("role-
|
|
586
|
+
return step_id.startswith(("role-model:", "role-models:", "role-models-custom:"))
|
|
576
587
|
|
|
577
588
|
|
|
578
589
|
def _discard_implicit_leader_selection(state: WizardState) -> None:
|
|
@@ -661,6 +672,7 @@ _FIELD_DEFAULTS: dict[str, Any] = {
|
|
|
661
672
|
"executor": "", "critic": "", "critic_pending_text": False,
|
|
662
673
|
"execution_identity_version": 1,
|
|
663
674
|
"role_counts": {}, "role_models": {}, "role_selection_order": [],
|
|
675
|
+
"role_models_custom": [],
|
|
664
676
|
"reuse_previous": None,
|
|
665
677
|
"use_defaults": None, "workers_override": "",
|
|
666
678
|
"workers_custom_pending": False,
|
|
@@ -129,6 +129,28 @@ def _next_phase_cell(raw: Any) -> str:
|
|
|
129
129
|
return f"{cell} ({pointer['status']})"
|
|
130
130
|
|
|
131
131
|
|
|
132
|
+
def _recommended_task_entry(remaining: list[dict], latest_key: str) -> dict | None:
|
|
133
|
+
"""남은 task 중 이 run 이 권하는 하나.
|
|
134
|
+
|
|
135
|
+
포인터가 `ready` 인 task 는 지금 바로 다음 phase 를 시작할 수 있는 task 다
|
|
136
|
+
— 그중 최신(카탈로그는 updatedAt 내림차순)을 권한다. `ready` 가 없으면
|
|
137
|
+
마지막으로 만진 task(`latest-task.json`)를, 그것도 목록에 없으면 최신
|
|
138
|
+
task 를 권한다. 실측(2026-09-09): 세 줄 전부가 `(추천)` 이었고 리드는
|
|
139
|
+
산문에서 2번을 권했다 — 추천이 여럿이면 라벨은 아무것도 고르지 않은
|
|
140
|
+
것이다.
|
|
141
|
+
"""
|
|
142
|
+
if not remaining:
|
|
143
|
+
return None
|
|
144
|
+
for entry in remaining:
|
|
145
|
+
pointer = next_phase.promote(entry.get("nextRecommendedPhase"))
|
|
146
|
+
if pointer["status"] == next_phase.STATUS_READY:
|
|
147
|
+
return entry
|
|
148
|
+
for entry in remaining:
|
|
149
|
+
if latest_key and entry.get("taskKey") == latest_key:
|
|
150
|
+
return entry
|
|
151
|
+
return remaining[0]
|
|
152
|
+
|
|
153
|
+
|
|
132
154
|
def _build_task_pick(state: WizardState) -> Prompt:
|
|
133
155
|
t = _p(state.workspace_root, "task_pick")
|
|
134
156
|
project_root = Path(state.project_root)
|
|
@@ -137,8 +159,14 @@ def _build_task_pick(state: WizardState) -> Prompt:
|
|
|
137
159
|
latest_key = latest.get("taskKey") or ""
|
|
138
160
|
latest_suffix = t["options"].get("_LATEST_SUFFIX", "")
|
|
139
161
|
remaining = [e for e in tasks if (e.get("workStatus") or "") != "done"]
|
|
162
|
+
# 추천은 하나이고 1번이다. 나머지는 최신순 그대로.
|
|
163
|
+
recommended = _recommended_task_entry(remaining, latest_key)
|
|
164
|
+
ordered = (
|
|
165
|
+
[recommended, *(e for e in remaining if e is not recommended)]
|
|
166
|
+
if recommended is not None else remaining
|
|
167
|
+
)
|
|
140
168
|
options: list[Option] = []
|
|
141
|
-
for entry in
|
|
169
|
+
for entry in ordered[:_recommendation_budget(state)]:
|
|
142
170
|
key = entry.get("taskKey") or ""
|
|
143
171
|
ttype = entry.get("taskType") or ""
|
|
144
172
|
# catalog entries are flat (render_task_catalog_discovery) — there is
|
|
@@ -147,7 +175,8 @@ def _build_task_pick(state: WizardState) -> Prompt:
|
|
|
147
175
|
nxt = _next_phase_cell(entry.get("nextRecommendedPhase"))
|
|
148
176
|
suffix = latest_suffix if key == latest_key else ""
|
|
149
177
|
label = f"{key} · {phase} · next: {nxt}{_contract_outcome_suffix(entry)}{suffix}"
|
|
150
|
-
options.append(_opt(value=key, label=label,
|
|
178
|
+
options.append(_opt(value=key, label=label,
|
|
179
|
+
recommended=entry is recommended))
|
|
151
180
|
for value, label in _static_options(t):
|
|
152
181
|
options.append(_opt(value=value, label=label))
|
|
153
182
|
return Prompt(step=S_TASK_PICK, kind="pick",
|
|
@@ -335,9 +364,11 @@ def _build_task_group(state: WizardState) -> Prompt:
|
|
|
335
364
|
t = _p(state.workspace_root, "task_group_no_suggestion")
|
|
336
365
|
recent_prefix = t.get("recent_label_prefix", "")
|
|
337
366
|
options: list[Option] = []
|
|
338
|
-
|
|
367
|
+
# 추천은 하나뿐이다: 가장 최근에 task·brief 활동이 있던 group — 새 task 는
|
|
368
|
+
# 진행 중인 작업 흐름에 속할 가능성이 가장 크다. 나머지는 후보다.
|
|
369
|
+
for index, tg in enumerate(recent):
|
|
339
370
|
options.append(_opt(f"{_RECENT_PREFIX}{tg}", f"{recent_prefix}{tg}",
|
|
340
|
-
recommended=
|
|
371
|
+
recommended=index == 0))
|
|
341
372
|
options.append(_opt(PICK_TYPE_CUSTOM, t["options"][PICK_TYPE_CUSTOM]))
|
|
342
373
|
return Prompt(
|
|
343
374
|
step=S_TASK_GROUP, kind="pick",
|
|
@@ -410,9 +441,11 @@ def _build_task_id(state: WizardState) -> Prompt:
|
|
|
410
441
|
t = _p(state.workspace_root, "task_id_no_suggestion")
|
|
411
442
|
recent_prefix = t.get("recent_label_prefix", "")
|
|
412
443
|
options: list[Option] = []
|
|
444
|
+
# 새 task 의 id 로 같은 group 의 기존 id 를 권할 근거는 없다 — 후보로만
|
|
445
|
+
# 싣고 추천은 비운다. 추천이 있는 경우는 brief frontmatter 의 제안뿐이고,
|
|
446
|
+
# 그것은 위의 `task_id_with_suggestion` 분기다.
|
|
413
447
|
for tid in recent:
|
|
414
|
-
options.append(_opt(f"{_RECENT_PREFIX}{tid}", f"{recent_prefix}{tid}"
|
|
415
|
-
recommended=True))
|
|
448
|
+
options.append(_opt(f"{_RECENT_PREFIX}{tid}", f"{recent_prefix}{tid}"))
|
|
416
449
|
options.append(_opt(PICK_TYPE_CUSTOM, t["options"][PICK_TYPE_CUSTOM]))
|
|
417
450
|
return Prompt(
|
|
418
451
|
step=S_TASK_ID, kind="pick",
|
|
@@ -713,12 +746,21 @@ def _build_brief_path_pick(state: WizardState) -> Prompt:
|
|
|
713
746
|
def add(value: str, label: str) -> None:
|
|
714
747
|
if len(options) >= budget:
|
|
715
748
|
return
|
|
716
|
-
|
|
749
|
+
# 첫 후보만 추천이다: 이 task 의 기존 brief → (있으면) 표준 경로의
|
|
750
|
+
# brief → 그룹 시작 순서상 첫 미착수 brief. 각각이 그 자리에서 가장
|
|
751
|
+
# 타당한 입력이고, 그 뒤 줄들은 대안이다.
|
|
752
|
+
options.append(_opt(value, label, recommended=not options))
|
|
717
753
|
|
|
718
754
|
if existing:
|
|
719
755
|
add("__existing__",
|
|
720
756
|
t["options"]["__existing__"].format(existing=existing))
|
|
721
|
-
|
|
757
|
+
# 표준 경로는 파일이 실제로 있을 때만 후보다. 없는 파일을 고르면
|
|
758
|
+
# `_require_file` 이 거절하므로, 그 줄은 선택지가 아니라 함정이고 추천은
|
|
759
|
+
# 더더욱 아니다.
|
|
760
|
+
if (
|
|
761
|
+
standard and standard != existing
|
|
762
|
+
and (Path(state.project_root) / standard).is_file()
|
|
763
|
+
):
|
|
722
764
|
add("__standard__",
|
|
723
765
|
t["options"]["__standard__"].format(standard=standard))
|
|
724
766
|
brief_label = t["labels"].get("brief_candidate", "{path}")
|
|
@@ -293,6 +293,7 @@ def _submit_reuse_previous(state: WizardState, value: str) -> Optional[str]:
|
|
|
293
293
|
conversion_payload.setdefault("kimiModel", state.kimi_model)
|
|
294
294
|
state.role_counts.clear()
|
|
295
295
|
state.role_models.clear()
|
|
296
|
+
state.role_models_custom.clear()
|
|
296
297
|
state.role_selection_order.clear()
|
|
297
298
|
_convert_v1_provider_selections(state, conversion_payload)
|
|
298
299
|
directive = inputs.get("directive")
|
|
@@ -9,6 +9,7 @@ from dataclasses import dataclass
|
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
from typing import Any, Iterable, Mapping, Sequence
|
|
11
11
|
|
|
12
|
+
from .convergence_reverify_prompt import RENDERED_BY_LINE
|
|
12
13
|
from .worker_prompt_body import analysis_worker_label
|
|
13
14
|
from .json_boundary import load_owned_object
|
|
14
15
|
from .worker_prompt_policy import (
|
|
@@ -313,6 +314,16 @@ def validate_reverify_prompt(
|
|
|
313
314
|
and boundary_position > first_heading.start()
|
|
314
315
|
):
|
|
315
316
|
errors.append("phase boundary block must precede reverify instructions")
|
|
317
|
+
if RENDERED_BY_LINE not in normalized[_task_instructions_offset(normalized):]:
|
|
318
|
+
# 손으로 쓴 지시문이 한 라운드를 버렸다(2026-09-09: `- Verdict:` 형식과 축약된
|
|
319
|
+
# 근거). 렌더러의 서명 줄이 없으면 그 지시문은 렌더러 출력이 아니다.
|
|
320
|
+
errors.append(
|
|
321
|
+
"reverify instruction is not the output of `okstra convergence "
|
|
322
|
+
"reverify-prompt` (missing the `**Rendered by:**` line) — render it with "
|
|
323
|
+
"`okstra convergence reverify-prompt --run-manifest <run-manifest> "
|
|
324
|
+
"--plan <round-plan.json> --worker <worker-id>` and pass that output "
|
|
325
|
+
"verbatim as --instruction; hand-written reverify instructions are refused"
|
|
326
|
+
)
|
|
316
327
|
errors.extend(_validate_output_contract_block(normalized))
|
|
317
328
|
return errors
|
|
318
329
|
|
|
@@ -61,7 +61,7 @@ The final `confirm` step is a normal `pick` step with three options — `Proceed
|
|
|
61
61
|
|
|
62
62
|
Never invent additional questions. **Never drop, hide, merge, reorder, or truncate** a `pick` / `pick_group` option — relay every `options[]` entry, including entries that carry a `(default)` / `(recommended)` suffix. Do not collapse a multi-option pick into a "recommended + Enter directly / Other" shortlist. The wizard's arrays are the complete authoritative choice sets, regardless of the current host UI's usual option limit. The run-prompt recommendation rule (1–2 recommendations + Enter directly) shapes the **option set** only for prompts this skill authors itself, never for wizard-provided options — you may not add, drop, or reorder a wizard option to produce a shortlist. It does not excuse you from recommending: before relaying a wizard step whose answer turns on something readable (the carried report, the sidecar the user already wrote, the prior Stage Map), read it, put what you found in the question body, and name which of the wizard's own options you recommend and why. Relaying a step with no context and no recommendation hands the whole question back to the user — see the lifecycle core contract "Asking the user (BLOCKING)".
|
|
63
63
|
|
|
64
|
-
**One recommendation, shown in one place.**
|
|
64
|
+
**One recommendation, shown in one place, and it is option 1.** The option carrying `recommended: true` is what this run computed, and the wizard already placed it first — append ` (추천)` to that option's label when you render it, and to no other. A checkbox step (`multi: true`) may flag several leading options: they are the recommended set, rendered the same way. Your prose recommendation names one of the flagged options. If you believe a different option is right, do not quietly recommend it in prose while the flagged one still reads as recommended on screen: say you disagree, name both, and let the user pick. When no option carries the flag, the run computed nothing for this step — recommend one from what you read and mark that one, and do not present the first option as a default just because it is first. **Enforced:** `scripts/okstra_ctl/wizard/state.py` `Prompt.__post_init__` refuses a step whose free-input option is not last, whose single-select recommendation is not exactly one option placed first (a checkbox step's recommended options must be its leading run), or that marks the free-input / abort escape as a recommendation.
|
|
65
65
|
|
|
66
66
|
## Step 1: Preflight
|
|
67
67
|
|
|
@@ -198,7 +198,7 @@ That is the entire interactive flow. The wizard handles:
|
|
|
198
198
|
- base-ref pick + git rev-parse validation (skipped when reusing an active worktree),
|
|
199
199
|
- `implementation`-only sub-flow: approved-plan path (frontmatter `approved: true` check) + stage pick (`auto` = the earliest incomplete stage whose dependencies are satisfied, or a specific stage number). Implementer slots use role-count / role-model like every other role (`executor` is only a compatibility alias for `implementer`). When an approved plan is selected and a `## PLAN DECISION` sidecar carrying `Status: approved`, exported from the report — matching the plan on source-report·seq — is detected in that run's sibling `user-responses/`, the approve-confirm step expands to 3 options (`yes_apply` recommended: approve + apply the option as exported / `yes` approve only / `no` abort) — `yes_apply` validates the option against the plan's `optionCandidates` before applying it via the existing approval·option path,
|
|
200
200
|
- `release-handoff`-only sub-flow: after the approved plan auto-resolves, a `handoff_stage_pick` multi-select — choose an eligible stage bundle (stage-group) or the whole task (when an accepted whole-task verification report exists); the result goes out as render-args' `stages` key (csv, empty when whole-task),
|
|
201
|
-
- launch selection after identity/worktree steps:
|
|
201
|
+
- launch selection after identity/worktree steps: one screen per static role, in profile order. A role that can run several instances (`max > 1`) is a checkbox step `role-models:<role>` (`multi: true`) — the number of models checked is the number of instances, there is no separate count question; the label states the profile range and recommended count, the recommended set is flagged, and the last row (`직접 선택`) opens `role-models-custom:<role>` with every executable candidate. An optional role (`min = 0`, e.g. critic) carries a `추가 안 함` row. A fixed single role (`min = max = 1`, e.g. report-writer) is a single pick `role-model:<role>:1`. current-session lead is this session and is listed on the confirmation summary, not as a wizard step. The wizard does not fork on defaults-vs-customize, does not show a provider roster multi-pick, and does not offer a separate implementer-provider pick. Dynamic verifiers are not chosen at launch. `--workers` is compatibility-only, not a launch picker. Repeated `--role-count` / `--role-model` tokens on `renderArgv` are intentional,
|
|
202
202
|
- **resume-clarification (in-session equivalent)** — there is no separate mode or flag matching the shell's `okstra.sh --resume-clarification`; two steps of the standard flow carry out its substance. (1) `reuse_previous` (yes/no to reuse the previous run's settings — in `requirements-discovery` / `error-analysis` / `implementation-planning`, only when prior run-inputs exist): YES prefills role-count·role-model·directive·related-tasks at once. (2) `clarification_pick`: if the **task-type's own** previous `final-report` exists it is auto-recommended as the carry-in input (falling back to the newest by mtime across all phases when absent), and the same run's `user-responses/` sidecar (answers the user filled in) is attached alongside. The chosen path is passed to prepare as `--clarification-response` — the user makes the sidecar via the report's `Export user response`, places it in `runs/<task-type>/user-responses/`, and re-runs the same phase,
|
|
203
203
|
- **re-verification scope (`reverify_scope_pick`, `implementation-planning` clarification re-runs only)** — asked right before `confirm` when the re-run is narrowable **or** an answered `C-NNN` traces to no stage. When every answered id traces to a stage: 3 options — `auto` (recommended — leave it to the lead's `okstra incremental-scope` decision) / `full` (re-verify every stage) / Enter directly (a stage-number CSV, validated against the prior report's Stage Map). When an id is unlinked, `auto` is omitted and the user names stages or picks `full`; that unlinked id does not freeze the run at full. The answer goes out as `--reverify-scope` and reaches the lead prompt as the `REVERIFY_SCOPE_MODE` / `REVERIFY_SCOPE_STAGES` tokens; it shapes that CLI's inputs rather than replacing the decision. The confirmation block's `reverify-scope` line names unlinked ids as needing stage numbers, not as a forced full re-run,
|
|
204
204
|
- `release-handoff` PR template override + persist scope,
|
|
@@ -42,6 +42,7 @@ from okstra_project.resolver import resolve_architecture # noqa: E402
|
|
|
42
42
|
|
|
43
43
|
from okstra_ctl.conformance import ( # noqa: E402
|
|
44
44
|
detect_surfaces,
|
|
45
|
+
exempt_stage_surface_conflicts,
|
|
45
46
|
evaluate_conformance,
|
|
46
47
|
manifest_required_surfaces,
|
|
47
48
|
missing_declared_scripts,
|
|
@@ -1888,6 +1889,19 @@ def _approved_plan_stage_entries(
|
|
|
1888
1889
|
return entries
|
|
1889
1890
|
|
|
1890
1891
|
|
|
1892
|
+
def _approved_plan_exempted_stages(stages: object) -> list[str]:
|
|
1893
|
+
"""`conformanceExemption` 을 선언한 stage 번호들(문자열, stageKey 접미 비교용).
|
|
1894
|
+
|
|
1895
|
+
`_approved_plan_stage_entries` 가 이미 형식을 판정한 뒤에만 부른다."""
|
|
1896
|
+
return sorted(
|
|
1897
|
+
str(stage.get("stage"))
|
|
1898
|
+
for stage in (stages if isinstance(stages, list) else [])
|
|
1899
|
+
if isinstance(stage, dict)
|
|
1900
|
+
and isinstance(stage.get("conformanceExemption"), str)
|
|
1901
|
+
and stage["conformanceExemption"].strip()
|
|
1902
|
+
)
|
|
1903
|
+
|
|
1904
|
+
|
|
1891
1905
|
def _record_malformed_plan_stages(data_path: Path, failures: list[str]) -> None:
|
|
1892
1906
|
failures.append(
|
|
1893
1907
|
f"conformance gate BLOCKING: approved plan evidence has malformed stages at {data_path}"
|
|
@@ -1979,7 +1993,12 @@ def _approved_plan_conformance_manifest(
|
|
|
1979
1993
|
planning = data.get("implementationPlanning") if isinstance(data, dict) else None
|
|
1980
1994
|
stages = planning.get("stages") if isinstance(planning, dict) else None
|
|
1981
1995
|
entries = _approved_plan_stage_entries(stages, data_path, failures)
|
|
1982
|
-
|
|
1996
|
+
if entries is None:
|
|
1997
|
+
return None
|
|
1998
|
+
return {
|
|
1999
|
+
"entries": entries,
|
|
2000
|
+
"exemptedStages": _approved_plan_exempted_stages(stages),
|
|
2001
|
+
}
|
|
1983
2002
|
|
|
1984
2003
|
|
|
1985
2004
|
def _declared_conformance_errors(
|
|
@@ -2022,11 +2041,21 @@ def _declared_conformance_errors(
|
|
|
2022
2041
|
)
|
|
2023
2042
|
if actual_capabilities != frozenset(declaration.get("requires") or []):
|
|
2024
2043
|
errors.append(f"stage {stage_number} requires mismatch")
|
|
2044
|
+
# 계획이 면제한 stage 에 구현이 실제 Tier 3 항목을 붙이는 것은 허용한다 —
|
|
2045
|
+
# 면제 stage 의 diff 가 db/io/http/external 표면을 건드려 diff-surface 대조에
|
|
2046
|
+
# 걸렸을 때, 승인된 계획은 불변이므로 그 항목이 유일한 진행 경로다
|
|
2047
|
+
# (2026-09-09 dev-10784 Stage 2). 계획에 없는 stage 의 항목은 여전히 거절한다.
|
|
2048
|
+
exempted_stage_numbers = {
|
|
2049
|
+
str(value) for value in declared_manifest.get("exemptedStages") or []
|
|
2050
|
+
}
|
|
2025
2051
|
for actual_entry in actual:
|
|
2026
2052
|
if not isinstance(actual_entry, dict):
|
|
2027
2053
|
continue
|
|
2028
2054
|
stage_number = str(actual_entry.get("stageKey") or "").rsplit("-stage-", 1)[-1]
|
|
2029
|
-
if
|
|
2055
|
+
if (
|
|
2056
|
+
stage_number not in declared_stage_numbers
|
|
2057
|
+
and stage_number not in exempted_stage_numbers
|
|
2058
|
+
):
|
|
2030
2059
|
errors.append(
|
|
2031
2060
|
f"stage {stage_number} actual manifest entry is not declared by approved plan"
|
|
2032
2061
|
)
|
|
@@ -2062,12 +2091,32 @@ def _planning_conformance_declarations(
|
|
|
2062
2091
|
return declarations
|
|
2063
2092
|
|
|
2064
2093
|
|
|
2065
|
-
def
|
|
2066
|
-
"""
|
|
2094
|
+
def _project_surface_patterns(project_root: Path) -> object:
|
|
2095
|
+
"""project.json `qaEnv.surfacePatterns` — 계획·구현 두 게이트가 같은 표를 쓴다."""
|
|
2096
|
+
path = project_json_path(project_root)
|
|
2097
|
+
if not path.is_file():
|
|
2098
|
+
return None
|
|
2099
|
+
try:
|
|
2100
|
+
return (json.loads(path.read_text()).get("qaEnv") or {}).get("surfacePatterns")
|
|
2101
|
+
except (OSError, json.JSONDecodeError):
|
|
2102
|
+
return None
|
|
2103
|
+
|
|
2104
|
+
|
|
2105
|
+
def _validate_planning_conformance_declared(
|
|
2106
|
+
report_path: Path,
|
|
2107
|
+
failures: list[str],
|
|
2108
|
+
surface_patterns: object = None,
|
|
2109
|
+
) -> None:
|
|
2110
|
+
"""계획 단계는 `Conformance tests:` / `Conformance exemption:` 선언 형식을 본다.
|
|
2067
2111
|
|
|
2068
2112
|
스크립트 파일과 `runCommand` 는 매칭 implementation stage 가 만든다.
|
|
2069
2113
|
선언만 있고 파일이 없는 것은 계획 게이트 실패가 아니다. 형식이 깨진
|
|
2070
2114
|
`conformanceTests` 는 여전히 실패한다.
|
|
2115
|
+
|
|
2116
|
+
면제 stage 의 `plannedPaths` 가 db/io/http/external 표면을 건드리면 여기서
|
|
2117
|
+
막는다 — 구현 게이트의 diff-surface 대조(`_validate_conformance_surfaces`)와
|
|
2118
|
+
같은 패턴이다. 종전에는 그 대조가 구현이 끝난 뒤에만 돌아, 승인된 계획을
|
|
2119
|
+
고칠 수 없는 자리에서 run 전체가 막혔다(2026-09-09 dev-10784 Stage 2).
|
|
2071
2120
|
"""
|
|
2072
2121
|
data_path = _data_path_for(report_path)
|
|
2073
2122
|
if not data_path.is_file():
|
|
@@ -2080,6 +2129,19 @@ def _validate_planning_conformance_declared(report_path: Path, failures: list[st
|
|
|
2080
2129
|
if not isinstance(ip, dict):
|
|
2081
2130
|
return
|
|
2082
2131
|
_planning_conformance_declarations(ip.get("stages"), failures)
|
|
2132
|
+
for conflict in exempt_stage_surface_conflicts(data, surface_patterns):
|
|
2133
|
+
failures.append(
|
|
2134
|
+
"conformance gate BLOCKING: stage "
|
|
2135
|
+
f"{conflict['stage']} declares `Conformance exemption:` but its "
|
|
2136
|
+
f"planned paths touch surface(s) {conflict['surfaces']}: "
|
|
2137
|
+
f"{', '.join(conflict['paths'])} — an exemption cannot hide a "
|
|
2138
|
+
"db/io/http/external change (prompts/profiles/implementation-planning.md "
|
|
2139
|
+
"\"Per-stage conformance declaration\"); declare `Conformance tests:` "
|
|
2140
|
+
f"with requires={conflict['surfaces']} for that stage, or move those "
|
|
2141
|
+
"paths out of it. The implementation run's diff-surface check "
|
|
2142
|
+
"blocks the same stage after the work is done, where the approved "
|
|
2143
|
+
"plan can no longer be corrected."
|
|
2144
|
+
)
|
|
2083
2145
|
|
|
2084
2146
|
|
|
2085
2147
|
def _validate_conformance_surfaces(
|
|
@@ -2100,7 +2162,10 @@ def _validate_conformance_surfaces(
|
|
|
2100
2162
|
"conformance gate BLOCKING: implementation diff touches undeclared "
|
|
2101
2163
|
f"surface(s) {sorted(uncovered)} — no in-scope stage declares "
|
|
2102
2164
|
"`requires` for them. Declare a conformance entry (requires=[...]) "
|
|
2103
|
-
"for the touching stage
|
|
2165
|
+
"for the touching stage — also when the approved plan exempted "
|
|
2166
|
+
"that stage: write the Tier 3 script under <task_root>/qa/scripts/ "
|
|
2167
|
+
"and its conformance-manifest.json entry with requires covering "
|
|
2168
|
+
"those surfaces; the approved plan's exemption is not rewritten. "
|
|
2104
2169
|
"(silent mock-green 방지 — DEV-9184)"
|
|
2105
2170
|
)
|
|
2106
2171
|
|
|
@@ -8108,8 +8173,8 @@ def _validate_translation_sidecar(
|
|
|
8108
8173
|
failures.append(
|
|
8109
8174
|
f"final-report has reportLanguage {lang!r} but no translation "
|
|
8110
8175
|
f"sidecar at {sidecar.name}. The human HTML rendered from the "
|
|
8111
|
-
"English source instead; re-run `report-finalize --only "
|
|
8112
|
-
"render-views`
|
|
8176
|
+
"English source instead; re-run `report-finalize --only translate "
|
|
8177
|
+
"--only render-views` to dispatch the translator and overlay it."
|
|
8113
8178
|
)
|
|
8114
8179
|
|
|
8115
8180
|
|
|
@@ -9398,17 +9463,10 @@ def main() -> int:
|
|
|
9398
9463
|
failures,
|
|
9399
9464
|
)
|
|
9400
9465
|
if task_type in ("implementation", "final-verification"):
|
|
9401
|
-
_sp = None
|
|
9402
|
-
_pj = project_json_path(project_root)
|
|
9403
|
-
if _pj.is_file():
|
|
9404
|
-
try:
|
|
9405
|
-
_sp = (json.loads(_pj.read_text()).get("qaEnv") or {}).get("surfacePatterns")
|
|
9406
|
-
except (OSError, json.JSONDecodeError):
|
|
9407
|
-
_sp = None
|
|
9408
9466
|
conformance_warnings = _validate_conformance(
|
|
9409
9467
|
report_path,
|
|
9410
9468
|
failures,
|
|
9411
|
-
surface_patterns=
|
|
9469
|
+
surface_patterns=_project_surface_patterns(project_root),
|
|
9412
9470
|
approved_plan_path=_approved_plan_path_from_run_inputs(
|
|
9413
9471
|
run_manifest_path,
|
|
9414
9472
|
failures,
|
|
@@ -9431,7 +9489,11 @@ def main() -> int:
|
|
|
9431
9489
|
)
|
|
9432
9490
|
if task_type in _END_STATE_PHASES:
|
|
9433
9491
|
if task_type == "implementation-planning":
|
|
9434
|
-
_validate_planning_conformance_declared(
|
|
9492
|
+
_validate_planning_conformance_declared(
|
|
9493
|
+
report_path,
|
|
9494
|
+
failures,
|
|
9495
|
+
surface_patterns=_project_surface_patterns(project_root),
|
|
9496
|
+
)
|
|
9435
9497
|
if not selected_direction_plan:
|
|
9436
9498
|
_validate_end_state_coverage(validation_data, brief_path, failures)
|
|
9437
9499
|
_validate_end_state_blocked_by(
|