okstra 0.176.1 → 0.177.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/dist/commands/execute/team.mjs +14 -4
  2. package/dist/commands/execute/team.mjs.map +1 -1
  3. package/dist/commands/lifecycle/install.mjs +0 -1
  4. package/dist/commands/lifecycle/install.mjs.map +1 -1
  5. package/docs/architecture.md +3 -3
  6. package/docs/cli.md +1 -1
  7. package/docs/project-structure-overview.md +2 -2
  8. package/docs/task-process/final-verification.md +5 -3
  9. package/package.json +1 -1
  10. package/runtime/BUILD.json +2 -2
  11. package/runtime/agents/workers/report-writer-worker.md +2 -2
  12. package/runtime/bin/okstra-compact-reminder.sh +2 -2
  13. package/runtime/bin/okstra-provider-exec.py +2 -7
  14. package/runtime/bin/okstra-render-report-views.py +13 -10
  15. package/runtime/prompts/coding-preflight/overview.md +2 -1
  16. package/runtime/prompts/lead/convergence.md +11 -6
  17. package/runtime/prompts/lead/okstra-lead-contract.md +3 -3
  18. package/runtime/prompts/lead/plan-body-verification.md +4 -2
  19. package/runtime/prompts/lead/report-writer.md +8 -4
  20. package/runtime/prompts/profiles/_common-contract.md +2 -2
  21. package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
  22. package/runtime/prompts/profiles/final-verification.md +11 -9
  23. package/runtime/prompts/profiles/improvement-discovery.md +1 -1
  24. package/runtime/prompts/profiles/release-handoff.md +7 -6
  25. package/runtime/prompts/wizard/prompts.ko.json +2 -2
  26. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -6
  27. package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +23 -1
  28. package/runtime/python/okstra_ctl/agent_invocation.py +17 -0
  29. package/runtime/python/okstra_ctl/agent_prompt_cli.py +15 -2
  30. package/runtime/python/okstra_ctl/dispatch_core.py +140 -17
  31. package/runtime/python/okstra_ctl/dispatch_state.py +60 -3
  32. package/runtime/python/okstra_ctl/execution_mutation_audit.py +49 -3
  33. package/runtime/python/okstra_ctl/handoff.py +27 -14
  34. package/runtime/python/okstra_ctl/model_cli.py +11 -2
  35. package/runtime/python/okstra_ctl/model_discovery.py +12 -0
  36. package/runtime/python/okstra_ctl/pane_reclaim.py +49 -43
  37. package/runtime/python/okstra_ctl/release_gate.py +56 -0
  38. package/runtime/python/okstra_ctl/render.py +53 -0
  39. package/runtime/python/okstra_ctl/report_contract.py +1 -0
  40. package/runtime/python/okstra_ctl/report_finalize.py +54 -0
  41. package/runtime/python/okstra_ctl/report_html/render.py +7 -4
  42. package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +4 -1
  43. package/runtime/python/okstra_ctl/run.py +119 -18
  44. package/runtime/python/okstra_ctl/stage_targets.py +73 -1
  45. package/runtime/python/okstra_ctl/team.py +84 -14
  46. package/runtime/python/okstra_ctl/tmux.py +2 -3
  47. package/runtime/python/okstra_ctl/wizard.py +19 -10
  48. package/runtime/python/okstra_ctl/worker_liveness.py +3 -1
  49. package/runtime/python/okstra_ctl/worker_runner.py +2 -2
  50. package/runtime/python/okstra_ctl/wrapper_status.py +15 -0
  51. package/runtime/python/okstra_ctl/write_policy.py +9 -1
  52. package/runtime/schemas/final-report-v2.0.schema.json +58 -19
  53. package/runtime/skills/okstra-run/SKILL.md +3 -3
  54. package/runtime/templates/reports/html/base.template.html +1 -2
  55. package/runtime/templates/reports/html/i18n/en.json +4 -0
  56. package/runtime/templates/reports/html/i18n/ko.json +4 -0
  57. package/runtime/templates/reports/html/tasks/final-verification.template.html +7 -1
  58. package/runtime/validators/validate-report-views.py +30 -17
  59. package/runtime/validators/validate-run.py +221 -90
  60. package/runtime/validators/validate_analysis_report.py +2 -5
  61. package/runtime/validators/validate_session_conformance.py +1 -1
  62. package/runtime/bin/okstra-trace-cleanup.sh +0 -185
@@ -51,8 +51,17 @@ from okstra_ctl.conformance import ( # noqa: E402
51
51
  qa_result_from_dict,
52
52
  validate_conformance_manifest,
53
53
  )
54
+ from okstra_ctl.dispatch_state import ( # noqa: E402
55
+ DispatchError,
56
+ v2_worker_state_key,
57
+ )
54
58
  from okstra_ctl.paths import RunRef # noqa: E402
55
59
  from okstra_ctl.report_contract import CURRENT_REPORT_SCHEMA_VERSION # noqa: E402
60
+ from okstra_ctl.release_gate import ( # noqa: E402
61
+ RELEASE_HANDOFF_TARGETS,
62
+ blocking_condition_ids,
63
+ release_handoff_allowed,
64
+ )
56
65
  from okstra_ctl.domain.host import HostNotRegistered # noqa: E402
57
66
  from okstra_ctl.registry.host_registry import default_host_registry # noqa: E402
58
67
  from okstra_ctl.build_tools import ( # noqa: E402
@@ -857,6 +866,12 @@ def extract_contract(
857
866
  required_worker_roles = []
858
867
  failures.append("requiredWorkerRoles is missing from run/task manifest")
859
868
 
869
+ optional_worker_roles = run_contract.get("optionalWorkerRoles")
870
+ if not isinstance(optional_worker_roles, list):
871
+ optional_worker_roles = task_contract.get("optionalWorkerRoles")
872
+ if not isinstance(optional_worker_roles, list):
873
+ optional_worker_roles = []
874
+
860
875
  lead_role = (
861
876
  run_contract.get("leadRole")
862
877
  or task_contract.get("leadRole")
@@ -887,6 +902,7 @@ def extract_contract(
887
902
  or ""
888
903
  ),
889
904
  "required_worker_roles": required_worker_roles,
905
+ "optional_worker_roles": optional_worker_roles,
890
906
  "required_agent_status_entries": [
891
907
  item
892
908
  for item in required_agent_status_entries
@@ -1211,6 +1227,24 @@ def _is_legal_concurrent_run_skip(
1211
1227
  )
1212
1228
 
1213
1229
 
1230
+ def _dispatch_roster_key(row: Mapping[str, Any]) -> str:
1231
+ """The roster worker this dispatch row started, v1 or v2.
1232
+
1233
+ A v2 row carries no `workerId` — `_validate_agent_dispatch_contract`
1234
+ fails the run when one is present, calling it a v1/v2 identity mix. So
1235
+ reading the roster key off that field alone left every v2 row invisible
1236
+ and reported workers okstra had in fact started as never dispatched. The
1237
+ v2 projection is the one dispatch itself uses.
1238
+ """
1239
+ worker_id = str(row.get("workerId", "")).strip()
1240
+ if worker_id:
1241
+ return worker_id
1242
+ try:
1243
+ return v2_worker_state_key(row)
1244
+ except DispatchError:
1245
+ return ""
1246
+
1247
+
1214
1248
  def _validate_cmux_workers_were_dispatched_by_okstra(
1215
1249
  team_state: dict,
1216
1250
  workers: list,
@@ -1238,9 +1272,11 @@ def _validate_cmux_workers_were_dispatched_by_okstra(
1238
1272
  if str(adapter.get("name", "")).strip() != "cmux":
1239
1273
  return
1240
1274
  recorded = {
1241
- str(row.get("workerId", "")).strip()
1275
+ key
1242
1276
  for row in team_state.get("workerDispatches") or []
1243
1277
  if isinstance(row, dict)
1278
+ for key in (_dispatch_roster_key(row),)
1279
+ if key
1244
1280
  }
1245
1281
  missing = []
1246
1282
  for worker in workers:
@@ -1442,7 +1478,15 @@ def validate_team_state(
1442
1478
  if status != "completed" and not reason:
1443
1479
  failures.append(f"{role} with status `{status}` must include a reason")
1444
1480
 
1445
- unexpected_roles = set(by_role) - set(expected_workers)
1481
+ # A declared optional role may appear in the roster and may equally be
1482
+ # absent: the run states which ones it can dispatch (today, the critics),
1483
+ # and running one is not a contract violation.
1484
+ optional_roles = {
1485
+ str(worker.get("role", "")).strip()
1486
+ for worker in contract.get("optional_worker_roles", [])
1487
+ if isinstance(worker, dict) and str(worker.get("role", "")).strip()
1488
+ }
1489
+ unexpected_roles = set(by_role) - set(expected_workers) - optional_roles
1446
1490
  for role in sorted(unexpected_roles):
1447
1491
  failures.append(f"unexpected worker role detected: {role}")
1448
1492
 
@@ -1616,60 +1660,46 @@ def _scan_token_usage_summary(
1616
1660
  # a section heading line (not as inline text inside a paragraph or table).
1617
1661
  _VERDICT_CARD_HEADING_RE = re.compile(r"^##[ \t]+Verdict Card\b", re.MULTILINE)
1618
1662
 
1619
- _V2_AI_HANDOFF_HEADINGS = (
1620
- "## AI Handoff Summary",
1621
- "## Clarification and User Decisions",
1622
- "## Evidence Ledger",
1623
- "## Task Deliverable:",
1624
- "## Cross Verification Audit",
1625
- "## Execution Audit",
1626
- "## Token and Cost Audit",
1627
- )
1628
-
1629
-
1630
- # `humanSummary` / `userNarrative` belong to the human HTML. The AI Markdown
1631
- # renders field names as headings or bold labels
1632
- # (scripts/okstra_ctl/report_markdown.py `humanise`), so this matches the two
1633
- # forms the renderer can emit rather than the raw JSON key — a key name in
1634
- # quotes no longer appears anywhere in the Markdown. Anchored to the start of a
1635
- # line so the same words inside worker prose are not a violation.
1636
- _V2_HUMAN_ONLY_RENDERED_RE = re.compile(
1637
- r"^(?:#{2,6}[ \t]+(?P<heading>Human Summary|User Narrative)[ \t]*$"
1638
- r"|[ \t]*-[ \t]+\*\*(?P<label>Human Summary|User Narrative)\*\*[ \t]*:)",
1639
- re.MULTILINE,
1640
- )
1641
-
1642
-
1643
- def _validate_v2_ai_handoff(content: str, failures: list[str]) -> None:
1644
- positions: list[int] = []
1645
- for heading in _V2_AI_HANDOFF_HEADINGS:
1646
- count = content.count(heading)
1647
- if count != 1:
1663
+ def _validate_v2_report(
1664
+ report_data: Mapping[str, Any],
1665
+ required_agent_status_entries: list[str],
1666
+ failures: list[str],
1667
+ ) -> None:
1668
+ """Contract checks for a schema-v2 report, read from its data.json.
1669
+
1670
+ The AI-handoff markdown is a deterministic rendering of this record, so the
1671
+ checks that used to scan it — heading count and order, human-only fields
1672
+ leaking into the AI artifact, a Reading Confirmation heading — were asking
1673
+ whether the renderer had done its job, not whether the report was sound.
1674
+ Schema enforcement in `render_final_report._enforce_schema` already answers
1675
+ the second question, and the first is settled by the template. What is left
1676
+ is the pair of facts the markdown could only ever carry second-hand: which
1677
+ agents the run must account for, and whether the token cells were filled.
1678
+
1679
+ Both are scanned over the serialized record, as text, because that is what
1680
+ the markdown scan was. Requiring each label to equal an
1681
+ `executionStatus[].role` would be a stricter rule than the contract states
1682
+ anywhere: the schema types `role` as a free string, no prompt or worker spec
1683
+ fixes its vocabulary, and `okstra_token_usage.report._match_worker_index`
1684
+ treats function-role spellings ("Analysis verifier", "Acceptance critic") as
1685
+ a shape reports do take — matching them by containment, never equality.
1686
+ Tightening this belongs with an authoring rule that says what to write, not
1687
+ on its own. Nothing is lost by reading the record instead of its rendering:
1688
+ the labels never come from the template or the i18n dictionaries.
1689
+ """
1690
+ serialized = json.dumps(report_data, ensure_ascii=False)
1691
+ for label in required_agent_status_entries:
1692
+ if label not in serialized:
1648
1693
  failures.append(
1649
- "schema-v2 AI handoff markdown requires exactly one "
1650
- f"`{heading}` heading; found {count}."
1694
+ f"final report does not include required agent status entry: {label}"
1695
+ )
1696
+ for placeholder in TOKEN_PLACEHOLDERS:
1697
+ if placeholder in serialized:
1698
+ failures.append(
1699
+ f"final report contains unsubstituted token placeholder `{placeholder}` — "
1700
+ "run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
1651
1701
  )
1652
- continue
1653
- positions.append(content.index(heading))
1654
- if len(positions) == len(_V2_AI_HANDOFF_HEADINGS) and positions != sorted(
1655
- positions
1656
- ):
1657
- failures.append(
1658
- "schema-v2 AI handoff markdown heading order does not match "
1659
- "templates/reports/final-report-v2.template.md."
1660
- )
1661
- for match in _V2_HUMAN_ONLY_RENDERED_RE.finditer(content):
1662
- field = match.group("heading") or match.group("label")
1663
- failures.append(
1664
- "schema-v2 AI handoff markdown contains human-only field "
1665
- f"{field!r}; render it only in the task-specific HTML."
1666
- )
1667
- if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1668
- failures.append(
1669
- "final report contains a `## 0. Reading Confirmation` heading — "
1670
- "Reading Confirmation lives in the worker audit sidecar, never "
1671
- "in the AI handoff markdown."
1672
- )
1702
+
1673
1703
 
1674
1704
  # Top-of-report Index block. The renderer
1675
1705
  # (scripts/okstra_ctl/render_final_report.py) injects `<a id="report-index">`
@@ -1777,16 +1807,34 @@ def _load_conformance_results(qa_dir: Path, manifest: dict) -> dict:
1777
1807
  return results
1778
1808
 
1779
1809
 
1780
- _DIFF_SUMMARY_RE = re.compile(r"^###\s+5\.7\.3\b.*?(?=^###\s|\Z)", re.MULTILINE | re.DOTALL)
1781
- _DIFF_ROW_PATH_RE = re.compile(r"^\|\s*`([^`]+)`\s*\|", re.MULTILINE)
1810
+ def _diff_summary_files(report_data: Mapping[str, Any]) -> list[str]:
1811
+ """implementation 리포트가 신고한 변경 파일 목록 (`implementation.diffSummary.files[].file`).
1782
1812
 
1813
+ 렌더된 §5.7.3 표를 정규식으로 긁던 자리다. 표는 data.json 의 이 배열에서
1814
+ 렌더되는 파생물이라, 표를 읽는 쪽은 렌더 형식이 바뀔 때마다 조용히 빈
1815
+ 목록을 돌려주고 — conformance / self-mock 두 게이트가 전부 통과로 열렸다.
1816
+ 스키마가 `implementation` 블록에서 `diffSummary` 를 required 로 잡고
1817
+ `rawStat` 이 비어있지 않으면 `files` 최소 1행을 요구하므로, 여기서는
1818
+ 구조가 어긋난 경우만 빈 목록으로 떨어뜨린다.
1783
1819
 
1784
- def _parse_diff_summary_files(content: str) -> list[str]:
1785
- """implementation 리포트 §5.7.3 Diff Summary 표의 첫 셀(백틱 경로)들을 추출."""
1786
- section = _DIFF_SUMMARY_RE.search(content)
1787
- if section is None:
1820
+ `diffSummary` 를 가진 task-type 은 implementation 뿐이다. final-verification
1821
+ 은 diff 를 `diffSummaryQuote` 문자열로만 인용하므로 두 게이트는 거기서
1822
+ (md 를 읽던 시절과 똑같이) vacuous 하다.
1823
+ """
1824
+ implementation = report_data.get("implementation")
1825
+ if not isinstance(implementation, Mapping):
1826
+ return []
1827
+ diff_summary = implementation.get("diffSummary")
1828
+ if not isinstance(diff_summary, Mapping):
1788
1829
  return []
1789
- return _DIFF_ROW_PATH_RE.findall(section.group(0))
1830
+ rows = diff_summary.get("files")
1831
+ if not isinstance(rows, list):
1832
+ return []
1833
+ return [
1834
+ row["file"]
1835
+ for row in rows
1836
+ if isinstance(row, Mapping) and isinstance(row.get("file"), str) and row["file"]
1837
+ ]
1790
1838
 
1791
1839
 
1792
1840
  _STAGE_RUN_DIR_RE = re.compile(r"^stage-\d+$")
@@ -2150,12 +2198,12 @@ def _validate_planning_conformance_declared(report_path: Path, failures: list[st
2150
2198
 
2151
2199
 
2152
2200
  def _validate_conformance_surfaces(
2153
- report_path: Path,
2201
+ report_data: Mapping[str, Any],
2154
2202
  scoped_manifest: dict,
2155
2203
  surface_patterns: object,
2156
2204
  failures: list[str],
2157
2205
  ) -> None:
2158
- changed_files = _parse_diff_summary_files(report_path.read_text(encoding="utf-8"))
2206
+ changed_files = _diff_summary_files(report_data)
2159
2207
  if not changed_files:
2160
2208
  return
2161
2209
  uncovered = (
@@ -2191,6 +2239,7 @@ def _validate_conformance(
2191
2239
  own stage suffix. Whole-task runs evaluate every entry.
2192
2240
  """
2193
2241
  warnings: list[str] = []
2242
+ report_data = _load_final_report_data(report_path)
2194
2243
  # conformance 산출물은 task-level(<task_root>/qa)에 있어 planning/
2195
2244
  # implementation/final-verification 가 공유한다. report_path 는
2196
2245
  # task_root/runs/<task-type>/reports/final-report.md (implementation 은
@@ -2231,7 +2280,7 @@ def _validate_conformance(
2231
2280
  f"{manifest_path} is absent"
2232
2281
  )
2233
2282
  _validate_conformance_surfaces(
2234
- report_path,
2283
+ report_data,
2235
2284
  empty_scoped_manifest,
2236
2285
  surface_patterns,
2237
2286
  failures,
@@ -2295,7 +2344,7 @@ def _validate_conformance(
2295
2344
  f"docs/superpowers/specs/2026-06-07-stage-conformance-qa-design.md."
2296
2345
  )
2297
2346
  _validate_conformance_surfaces(
2298
- report_path,
2347
+ report_data,
2299
2348
  scoped,
2300
2349
  surface_patterns,
2301
2350
  failures,
@@ -2633,13 +2682,13 @@ def _validate_selfmock(report_path: Path, failures: list[str]) -> None:
2633
2682
  Stage-isolated runs read their own `self-mock-stage-<N>.json`; whole-task runs
2634
2683
  read the flat `self-mock.json`.
2635
2684
 
2636
- Only the implementation template renders the §5.7.3 diff summary
2637
- (`templates/reports/final-report-v2.template.md:589`); a final-verification report
2638
- quotes the diff as a blockquote instead, so this gate is vacuous there by
2639
- design — self-mock is enforced at the implementation stage, and
2640
- final-verification is a read-only re-verify that adds no test files.
2685
+ Only an implementation report carries `implementation.diffSummary.files[]`;
2686
+ a final-verification report records the diff as the `diffSummaryQuote`
2687
+ string instead, so this gate is vacuous there by design — self-mock is
2688
+ enforced at the implementation stage, and final-verification is a read-only
2689
+ re-verify that adds no test files.
2641
2690
  """
2642
- changed = _parse_diff_summary_files(report_path.read_text(encoding="utf-8"))
2691
+ changed = _diff_summary_files(_load_final_report_data(report_path))
2643
2692
  test_files = [
2644
2693
  path
2645
2694
  for path in changed
@@ -2740,8 +2789,8 @@ def _check_selfmock_changed_files(
2740
2789
  `--changed-file` is what the mutation adapters select their production
2741
2790
  sources from. Omit it and every adapter gets an empty target set, which is
2742
2791
  not a finding but reads like one had been looked for. Requiring the sidecar
2743
- to account for every file in the §5.7.3 diff summary keeps "gate B saw this
2744
- stage" apart from "gate B was handed nothing".
2792
+ to account for every file in `implementation.diffSummary.files[]` keeps
2793
+ "gate B saw this stage" apart from "gate B was handed nothing".
2745
2794
 
2746
2795
  Coverage is asked of the WHOLE diff, not just its test files: the production
2747
2796
  sources are precisely the part gate A never looks at.
@@ -2766,7 +2815,8 @@ def _check_selfmock_changed_files(
2766
2815
  f"the mutation gate — {sidecar} records changedFiles={declared}. Each "
2767
2816
  "adapter picks its production sources out of that set, so a file left "
2768
2817
  "out is a file no mutant was ever generated for. Pass every path from "
2769
- "the §5.7.3 diff summary with `--changed-file`, production sources "
2818
+ "the report's `implementation.diffSummary.files[]` with `--changed-file`, "
2819
+ "production sources "
2770
2820
  "included — filtering to the test files leaves gate B nothing to run "
2771
2821
  "on and reports a pass it never earned."
2772
2822
  )
@@ -2830,6 +2880,10 @@ def validate_report(
2830
2880
  failures.append(f"final report is missing: {report_path}")
2831
2881
  return
2832
2882
 
2883
+ if (report_data or {}).get("schemaVersion") == "2.0":
2884
+ _validate_v2_report(report_data or {}, required_agent_status_entries, failures)
2885
+ return
2886
+
2833
2887
  content = report_path.read_text()
2834
2888
  for label in required_agent_status_entries:
2835
2889
  if label not in content:
@@ -2844,10 +2898,6 @@ def validate_report(
2844
2898
  "run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
2845
2899
  )
2846
2900
 
2847
- if (report_data or {}).get("schemaVersion") == "2.0":
2848
- _validate_v2_ai_handoff(content, failures)
2849
- return
2850
-
2851
2901
  # Catch the "workers typed `0` / `pending` instead of the placeholder"
2852
2902
  # failure mode that bypasses the placeholder check above.
2853
2903
  _scan_token_usage_summary(
@@ -6247,7 +6297,7 @@ def _validate_verified_row_recorded(
6247
6297
  report_path: Path,
6248
6298
  failures: list[str],
6249
6299
  ) -> None:
6250
- """An accepted single-stage verification must leave its `verified` row.
6300
+ """A release-ready single-stage verification must leave its `verified` row.
6251
6301
 
6252
6302
  `okstra handoff record-verified` validates its own inputs, but nothing
6253
6303
  checked that it ever ran. Skipping it leaves the report saying `accepted`
@@ -6258,8 +6308,7 @@ def _validate_verified_row_recorded(
6258
6308
  """
6259
6309
  if str(data.get("verificationScope") or "") != "single-stage":
6260
6310
  return
6261
- token = str((data.get("finalVerdict") or {}).get("verdictToken") or "").strip()
6262
- if token.lower() != "accepted":
6311
+ if not release_handoff_allowed(data):
6263
6312
  return
6264
6313
  stages = {
6265
6314
  row.get("stage")
@@ -6279,7 +6328,7 @@ def _validate_verified_row_recorded(
6279
6328
  missing = sorted(s for s in stages if s not in verified)
6280
6329
  if missing:
6281
6330
  failures.append(
6282
- f"final-verification accepted stage(s) {missing} but "
6331
+ f"final-verification cleared stage(s) {missing} for release but "
6283
6332
  "`runs/implementation-planning/consumers.jsonl` carries no "
6284
6333
  "`verified` row for them. Run `okstra handoff record-verified` "
6285
6334
  "before finishing — without that row the report says accepted "
@@ -6308,11 +6357,11 @@ def _validate_verifier_fail_blocks_verdict(data: dict, failures: list[str]) -> N
6308
6357
  })
6309
6358
  if not failed:
6310
6359
  return
6311
- token = str((data.get("verdictCard") or {}).get("verdictToken") or "").strip()
6360
+ token = str((data.get("finalVerdict") or {}).get("verdictToken") or "").strip()
6312
6361
  if token in _PASSING_VERDICT_TOKENS:
6313
6362
  failures.append(
6314
6363
  f"final-report data.json: verifier(s) {failed} recorded "
6315
- f"`verdict: FAIL` but `verdictCard.verdictToken` is `{token}`. A "
6364
+ f"`verdict: FAIL` but `finalVerdict.verdictToken` is `{token}`. A "
6316
6365
  "verifier rejection MUST survive into the published verdict — "
6317
6366
  "dropping it during synthesis is how rejected work reaches "
6318
6367
  "`release-handoff`. Either carry the FAIL into a blocking verdict "
@@ -7914,6 +7963,82 @@ def _validate_stage_has_requirement(data: dict, failures: list[str]) -> None:
7914
7963
  )
7915
7964
 
7916
7965
 
7966
+ _ADDED_SURFACE_NO_CALLER_RE = re.compile(r"^\s*none\b", re.IGNORECASE)
7967
+
7968
+
7969
+ def _validate_added_surface_audit(data: dict, failures: list[str]) -> None:
7970
+ """Every surface the diff added is traced to a requirement, exempted, or paid for.
7971
+
7972
+ The coverage table proves each requirement reached the diff. Nothing proved
7973
+ the reverse — that each thing the diff added answers a requirement — so work
7974
+ nobody asked for passed every gate. This check reads the reverse table and
7975
+ refuses a row that calls itself over-delivery without the blocker or
7976
+ condition it became: a caller-less surface is an acceptance blocker, and a
7977
+ surface with callers but no requirement is a conditional-acceptance
7978
+ condition (ADR-0009 grades the two differently on purpose).
7979
+ """
7980
+ fv = data.get("finalVerification")
7981
+ if not isinstance(fv, Mapping):
7982
+ return
7983
+ rows = fv.get("addedSurfaceAudit")
7984
+ if not isinstance(rows, list):
7985
+ return
7986
+ blocker_ids = {
7987
+ str(row.get("id"))
7988
+ for row in (fv.get("acceptanceBlockers") or [])
7989
+ if isinstance(row, Mapping)
7990
+ }
7991
+ condition_ids = {
7992
+ str(row.get("id"))
7993
+ for row in ((data.get("finalVerdict") or {}).get(
7994
+ "conditionalAcceptanceConditions") or [])
7995
+ if isinstance(row, Mapping)
7996
+ }
7997
+ for row in rows:
7998
+ if not isinstance(row, Mapping):
7999
+ continue
8000
+ row_id = str(row.get("id") or "<id 없음>")
8001
+ disposition = str(row.get("disposition") or "")
8002
+ note = str(row.get("note") or "")
8003
+ if disposition == "traced" and not str(row.get("requirement") or "").strip():
8004
+ failures.append(
8005
+ f"final-verification: addedSurfaceAudit {row_id} is `traced` but "
8006
+ "names no requirement — a surface is traced to something the "
8007
+ "brief asked for, or it is not traced."
8008
+ )
8009
+ continue
8010
+ if disposition != "over-delivery":
8011
+ continue
8012
+ caller_less = bool(
8013
+ _ADDED_SURFACE_NO_CALLER_RE.match(str(row.get("callers") or ""))
8014
+ )
8015
+ expected, known = (
8016
+ ("AB", blocker_ids) if caller_less else ("CA", condition_ids)
8017
+ )
8018
+ cited = set(re.findall(rf"\b{expected}-\d{{3,}}\b", note))
8019
+ if not cited:
8020
+ failures.append(
8021
+ f"final-verification: addedSurfaceAudit {row_id} is "
8022
+ f"`over-delivery` with callers "
8023
+ f"{'none' if caller_less else 'recorded'}, so its note MUST cite "
8024
+ f"the `{expected}-NNN` row it became — "
8025
+ + (
8026
+ "a caller-less surface is an acceptance blocker"
8027
+ if caller_less
8028
+ else "a surface with callers but no requirement is a "
8029
+ "conditional-acceptance condition"
8030
+ )
8031
+ + "."
8032
+ )
8033
+ continue
8034
+ missing = sorted(cited - known)
8035
+ if missing:
8036
+ failures.append(
8037
+ f"final-verification: addedSurfaceAudit {row_id} cites "
8038
+ f"{missing}, which the report does not carry."
8039
+ )
8040
+
8041
+
7917
8042
  def _validate_final_verification_consistency(data: dict, failures: list[str]) -> None:
7918
8043
  """Enforce verdict ↔ blocker/condition/routing consistency on the
7919
8044
  final-verification data.json (SSOT). The schema guarantees field SHAPE;
@@ -7923,6 +8048,7 @@ def _validate_final_verification_consistency(data: dict, failures: list[str]) ->
7923
8048
  """
7924
8049
  if (data.get("header") or {}).get("taskType") != "final-verification":
7925
8050
  return
8051
+ _validate_added_surface_audit(data, failures)
7926
8052
  verdict = data.get("finalVerdict") or {}
7927
8053
  token = (verdict.get("verdictToken") or "").strip().lower()
7928
8054
  fv = data.get("finalVerification") or {}
@@ -7954,14 +8080,19 @@ def _validate_final_verification_consistency(data: dict, failures: list[str]) ->
7954
8080
  "final-verification: verdict `conditional-accept` but "
7955
8081
  "conditionalAcceptanceConditions is empty — list every condition."
7956
8082
  )
7957
- if (
7958
- routing_token in {"release-handoff", "release-handoff(stage-group)"}
7959
- and token != "accepted"
7960
- ):
8083
+ if routing_token in RELEASE_HANDOFF_TARGETS and not release_handoff_allowed(data):
8084
+ blocking = blocking_condition_ids(data)
8085
+ reason = (
8086
+ f"condition(s) {blocking} declare `blocksReleaseHandoff: true` "
8087
+ "(a condition with no declaration counts as blocking)"
8088
+ if blocking
8089
+ else f"verdict is `{token}`"
8090
+ )
7961
8091
  failures.append(
7962
8092
  f"final-verification: routingRecommendation cites `release-handoff` "
7963
- f"but verdict is `{token}` — release-handoff routing is allowed only "
7964
- "when the verdict is `accepted`."
8093
+ f"but {reason} — release-handoff routing needs an `accepted` verdict, "
8094
+ "or a `conditional-accept` whose every condition declares "
8095
+ "`blocksReleaseHandoff: false`."
7965
8096
  )
7966
8097
 
7967
8098
  scope = data.get("verificationScope", "whole-task")
@@ -422,11 +422,8 @@ def _validate_analysis_verdict(
422
422
  expected = "blocked" if worker_blocked or scope_blocked else (
423
423
  "analysis-partial" if partial else "analysis-complete"
424
424
  )
425
- actual = {
426
- str((data.get("verdictCard") or {}).get("verdictToken") or ""),
427
- str((data.get("finalVerdict") or {}).get("verdictToken") or ""),
428
- }
429
- if actual == {expected}:
425
+ actual = str((data.get("finalVerdict") or {}).get("verdictToken") or "")
426
+ if actual == expected:
430
427
  return
431
428
  reasons = []
432
429
  if worker_blocked:
@@ -1034,7 +1034,7 @@ def _check_batch_cleanup_checkpoints(
1034
1034
  실제로 일어났는지 (prompts/profiles/_common-contract.md 'Phase-start cleanup').
1035
1035
  R1=convergence round 1 직전, R2=report-writer dispatch 직전(수렴이 있었으면
1036
1036
  마지막 라운드 이후 별도 1회). ISO-8601 ts 는 lexicographic 비교가 곧 시간순."""
1037
- detail = "prompts/profiles/_common-contract.md 'Phase-start cleanup'"
1037
+ detail = "prompts/lead/okstra-lead-contract.md 'Run-scoped worker-resource lifecycle'"
1038
1038
  cleanup_ts = sorted(ts for ts, _line in by_phase.get("phase-batch-cleanup", []))
1039
1039
  conv_ts = sorted(ts for ts, _line in by_phase.get("phase-5.5-convergence", []))
1040
1040
  synth_ts = sorted(ts for ts, _line in by_phase.get("phase-6-synthesis", []))