okstra 0.177.0 → 0.178.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/dist/commands/execute/team.mjs +14 -4
  2. package/dist/commands/execute/team.mjs.map +1 -1
  3. package/dist/commands/lifecycle/install.mjs +0 -1
  4. package/dist/commands/lifecycle/install.mjs.map +1 -1
  5. package/docs/architecture.md +3 -3
  6. package/docs/cli.md +1 -1
  7. package/docs/project-structure-overview.md +2 -2
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/workers/report-writer-worker.md +1 -1
  11. package/runtime/bin/okstra-compact-reminder.sh +2 -2
  12. package/runtime/bin/okstra-render-report-views.py +13 -10
  13. package/runtime/prompts/lead/okstra-lead-contract.md +2 -2
  14. package/runtime/prompts/lead/report-writer.md +5 -1
  15. package/runtime/prompts/profiles/_common-contract.md +1 -1
  16. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -6
  17. package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +23 -1
  18. package/runtime/python/okstra_ctl/agent_invocation.py +17 -0
  19. package/runtime/python/okstra_ctl/agent_prompt_cli.py +13 -0
  20. package/runtime/python/okstra_ctl/dispatch_core.py +96 -17
  21. package/runtime/python/okstra_ctl/dispatch_state.py +57 -0
  22. package/runtime/python/okstra_ctl/model_cli.py +11 -2
  23. package/runtime/python/okstra_ctl/model_discovery.py +12 -0
  24. package/runtime/python/okstra_ctl/pane_reclaim.py +49 -43
  25. package/runtime/python/okstra_ctl/render.py +53 -0
  26. package/runtime/python/okstra_ctl/report_html/render.py +7 -4
  27. package/runtime/python/okstra_ctl/report_views.py +35 -0
  28. package/runtime/python/okstra_ctl/run.py +88 -2
  29. package/runtime/python/okstra_ctl/team.py +84 -14
  30. package/runtime/python/okstra_ctl/tmux.py +2 -3
  31. package/runtime/python/okstra_ctl/user_response.py +20 -4
  32. package/runtime/python/okstra_ctl/worker_runner.py +2 -2
  33. package/runtime/python/okstra_ctl/write_policy.py +9 -1
  34. package/runtime/schemas/final-report-v2.0.schema.json +24 -1
  35. package/runtime/skills/okstra-run/SKILL.md +3 -3
  36. package/runtime/templates/reports/final-report-v2.template.md +2 -1
  37. package/runtime/templates/reports/html/base.template.html +1 -2
  38. package/runtime/validators/validate-report-views.py +30 -17
  39. package/runtime/validators/validate-run.py +256 -78
  40. package/runtime/validators/validate_session_conformance.py +1 -1
  41. package/runtime/bin/okstra-trace-cleanup.sh +0 -185
@@ -2,20 +2,26 @@
2
2
  """Validate the self-contained HTML view produced by Phase 7 step 1.5
3
3
  (``scripts/okstra-render-report-views.py``).
4
4
 
5
- Checks, for a given final-report MD path:
6
- 1. ``*.html`` sibling exists when the report has clarification rows,
7
- an analysis-review contract, or a plan-approval contract.
8
- 2. HTML's ``source-sha256`` in run-meta matches the current MD body —
9
- stale html detection.
10
- 3. HTML's §5.6 / §5.7 / §5.8 deliverable regions contain no
11
- ``<textarea>`` / ``<input>`` / ``<select>`` (form-attach is
12
- restricted to §1 clarification rows).
13
- 4. HTML has no external URLs in ``<script src=>`` / ``<link href=>`` /
14
- ``<img src=>`` — self-contained guarantee.
15
- 5. Every Response ID in HTML matches the §1 Clarification Items table
16
- of the source MD (1:1).
17
- 6. Analysis reports contain all three Analysis Review decisions and an
18
- affected-ID selector exactly matching the structured IDs in data.json.
5
+ Takes the report's Markdown path as its handle and branches on schema.
6
+
7
+ A schema-v2 report is checked against its ``.data.json`` — the record the
8
+ HTML was rendered from — and never against the Markdown sibling, which is a
9
+ second rendering of that same record:
10
+ 1. the ``*.html`` sibling exists;
11
+ 2. the document carries the task type's own template;
12
+ 3. every human field the task type requires is present, and no audit field
13
+ leaked into the human main;
14
+ 4. each visualization's svg node ids match its fallback ids;
15
+ 5. the Response IDs in the HTML match ``clarificationItems[]`` 1:1;
16
+ 6. no external URL appears in ``<script src=>`` / ``<link href=>`` /
17
+ ``<img src=>`` — the self-contained guarantee.
18
+
19
+ A schema-v1 report keeps the Markdown as its source, so its checks read the
20
+ §1 Clarification Items table (fail-closed when the heading is present but the
21
+ table will not parse), require the HTML sibling only when the report carries
22
+ clarification rows or an analysis-review / plan-approval contract, forbid form
23
+ controls in the §5.6 / §5.7 / §5.8 deliverable regions, and enforce the same
24
+ Response-ID parity and self-contained rules.
19
25
 
20
26
  Exit codes: 0 on success, 1 on any failure. Failures are printed one
21
27
  per line to stderr.
@@ -179,14 +185,21 @@ def _validate_analysis_review_html(
179
185
 
180
186
 
181
187
  def validate(report_path: Path) -> list[str]:
188
+ """Validate the html view of the report named by *report_path*.
189
+
190
+ A schema-v2 report is checked against its data.json and the rendered html;
191
+ the AI-handoff markdown is a sibling rendering of the same record and is
192
+ never read here, so it does not have to exist. Schema-v1 reports keep the
193
+ markdown as their source and still require it.
194
+ """
182
195
  failures: list[str] = []
196
+ v2 = _load_v2_data(report_path)
197
+ if v2 is not None:
198
+ return _validate_v2(report_path, v2[0], v2[1])
183
199
  if not report_path.is_file():
184
200
  return [f"final-report not found: {report_path}"]
185
201
 
186
202
  md = report_path.read_text(encoding="utf-8")
187
- v2 = _load_v2_data(report_path)
188
- if v2 is not None:
189
- return _validate_v2(report_path, v2[0], v2[1])
190
203
  html_path = html_view_path(report_path)
191
204
  # §1 헤딩이 있는데 파싱이 실패하면 md_ids 가 빈 []이 되어 "clarification 없음
192
205
  # → skip" 으로 흘러 HTML form parity 게이트가 조용히 열린다. fail-closed.
@@ -51,6 +51,10 @@ from okstra_ctl.conformance import ( # noqa: E402
51
51
  qa_result_from_dict,
52
52
  validate_conformance_manifest,
53
53
  )
54
+ from okstra_ctl.dispatch_state import ( # noqa: E402
55
+ DispatchError,
56
+ v2_worker_state_key,
57
+ )
54
58
  from okstra_ctl.paths import RunRef # noqa: E402
55
59
  from okstra_ctl.report_contract import CURRENT_REPORT_SCHEMA_VERSION # noqa: E402
56
60
  from okstra_ctl.release_gate import ( # noqa: E402
@@ -862,6 +866,12 @@ def extract_contract(
862
866
  required_worker_roles = []
863
867
  failures.append("requiredWorkerRoles is missing from run/task manifest")
864
868
 
869
+ optional_worker_roles = run_contract.get("optionalWorkerRoles")
870
+ if not isinstance(optional_worker_roles, list):
871
+ optional_worker_roles = task_contract.get("optionalWorkerRoles")
872
+ if not isinstance(optional_worker_roles, list):
873
+ optional_worker_roles = []
874
+
865
875
  lead_role = (
866
876
  run_contract.get("leadRole")
867
877
  or task_contract.get("leadRole")
@@ -892,6 +902,7 @@ def extract_contract(
892
902
  or ""
893
903
  ),
894
904
  "required_worker_roles": required_worker_roles,
905
+ "optional_worker_roles": optional_worker_roles,
895
906
  "required_agent_status_entries": [
896
907
  item
897
908
  for item in required_agent_status_entries
@@ -1216,6 +1227,24 @@ def _is_legal_concurrent_run_skip(
1216
1227
  )
1217
1228
 
1218
1229
 
1230
+ def _dispatch_roster_key(row: Mapping[str, Any]) -> str:
1231
+ """The roster worker this dispatch row started, v1 or v2.
1232
+
1233
+ A v2 row carries no `workerId` — `_validate_agent_dispatch_contract`
1234
+ fails the run when one is present, calling it a v1/v2 identity mix. So
1235
+ reading the roster key off that field alone left every v2 row invisible
1236
+ and reported workers okstra had in fact started as never dispatched. The
1237
+ v2 projection is the one dispatch itself uses.
1238
+ """
1239
+ worker_id = str(row.get("workerId", "")).strip()
1240
+ if worker_id:
1241
+ return worker_id
1242
+ try:
1243
+ return v2_worker_state_key(row)
1244
+ except DispatchError:
1245
+ return ""
1246
+
1247
+
1219
1248
  def _validate_cmux_workers_were_dispatched_by_okstra(
1220
1249
  team_state: dict,
1221
1250
  workers: list,
@@ -1243,9 +1272,11 @@ def _validate_cmux_workers_were_dispatched_by_okstra(
1243
1272
  if str(adapter.get("name", "")).strip() != "cmux":
1244
1273
  return
1245
1274
  recorded = {
1246
- str(row.get("workerId", "")).strip()
1275
+ key
1247
1276
  for row in team_state.get("workerDispatches") or []
1248
1277
  if isinstance(row, dict)
1278
+ for key in (_dispatch_roster_key(row),)
1279
+ if key
1249
1280
  }
1250
1281
  missing = []
1251
1282
  for worker in workers:
@@ -1447,7 +1478,15 @@ def validate_team_state(
1447
1478
  if status != "completed" and not reason:
1448
1479
  failures.append(f"{role} with status `{status}` must include a reason")
1449
1480
 
1450
- unexpected_roles = set(by_role) - set(expected_workers)
1481
+ # A declared optional role may appear in the roster and may equally be
1482
+ # absent: the run states which ones it can dispatch (today, the critics),
1483
+ # and running one is not a contract violation.
1484
+ optional_roles = {
1485
+ str(worker.get("role", "")).strip()
1486
+ for worker in contract.get("optional_worker_roles", [])
1487
+ if isinstance(worker, dict) and str(worker.get("role", "")).strip()
1488
+ }
1489
+ unexpected_roles = set(by_role) - set(expected_workers) - optional_roles
1451
1490
  for role in sorted(unexpected_roles):
1452
1491
  failures.append(f"unexpected worker role detected: {role}")
1453
1492
 
@@ -1621,60 +1660,46 @@ def _scan_token_usage_summary(
1621
1660
  # a section heading line (not as inline text inside a paragraph or table).
1622
1661
  _VERDICT_CARD_HEADING_RE = re.compile(r"^##[ \t]+Verdict Card\b", re.MULTILINE)
1623
1662
 
1624
- _V2_AI_HANDOFF_HEADINGS = (
1625
- "## AI Handoff Summary",
1626
- "## Clarification and User Decisions",
1627
- "## Evidence Ledger",
1628
- "## Task Deliverable:",
1629
- "## Cross Verification Audit",
1630
- "## Execution Audit",
1631
- "## Token and Cost Audit",
1632
- )
1633
-
1634
-
1635
- # `humanSummary` / `userNarrative` belong to the human HTML. The AI Markdown
1636
- # renders field names as headings or bold labels
1637
- # (scripts/okstra_ctl/report_markdown.py `humanise`), so this matches the two
1638
- # forms the renderer can emit rather than the raw JSON key — a key name in
1639
- # quotes no longer appears anywhere in the Markdown. Anchored to the start of a
1640
- # line so the same words inside worker prose are not a violation.
1641
- _V2_HUMAN_ONLY_RENDERED_RE = re.compile(
1642
- r"^(?:#{2,6}[ \t]+(?P<heading>Human Summary|User Narrative)[ \t]*$"
1643
- r"|[ \t]*-[ \t]+\*\*(?P<label>Human Summary|User Narrative)\*\*[ \t]*:)",
1644
- re.MULTILINE,
1645
- )
1646
-
1647
-
1648
- def _validate_v2_ai_handoff(content: str, failures: list[str]) -> None:
1649
- positions: list[int] = []
1650
- for heading in _V2_AI_HANDOFF_HEADINGS:
1651
- count = content.count(heading)
1652
- if count != 1:
1663
+ def _validate_v2_report(
1664
+ report_data: Mapping[str, Any],
1665
+ required_agent_status_entries: list[str],
1666
+ failures: list[str],
1667
+ ) -> None:
1668
+ """Contract checks for a schema-v2 report, read from its data.json.
1669
+
1670
+ The AI-handoff markdown is a deterministic rendering of this record, so the
1671
+ checks that used to scan it — heading count and order, human-only fields
1672
+ leaking into the AI artifact, a Reading Confirmation heading — were asking
1673
+ whether the renderer had done its job, not whether the report was sound.
1674
+ Schema enforcement in `render_final_report._enforce_schema` already answers
1675
+ the second question, and the first is settled by the template. What is left
1676
+ is the pair of facts the markdown could only ever carry second-hand: which
1677
+ agents the run must account for, and whether the token cells were filled.
1678
+
1679
+ Both are scanned over the serialized record, as text, because that is what
1680
+ the markdown scan was. Requiring each label to equal an
1681
+ `executionStatus[].role` would be a stricter rule than the contract states
1682
+ anywhere: the schema types `role` as a free string, no prompt or worker spec
1683
+ fixes its vocabulary, and `okstra_token_usage.report._match_worker_index`
1684
+ treats function-role spellings ("Analysis verifier", "Acceptance critic") as
1685
+ a shape reports do take — matching them by containment, never equality.
1686
+ Tightening this belongs with an authoring rule that says what to write, not
1687
+ on its own. Nothing is lost by reading the record instead of its rendering:
1688
+ the labels never come from the template or the i18n dictionaries.
1689
+ """
1690
+ serialized = json.dumps(report_data, ensure_ascii=False)
1691
+ for label in required_agent_status_entries:
1692
+ if label not in serialized:
1653
1693
  failures.append(
1654
- "schema-v2 AI handoff markdown requires exactly one "
1655
- f"`{heading}` heading; found {count}."
1694
+ f"final report does not include required agent status entry: {label}"
1695
+ )
1696
+ for placeholder in TOKEN_PLACEHOLDERS:
1697
+ if placeholder in serialized:
1698
+ failures.append(
1699
+ f"final report contains unsubstituted token placeholder `{placeholder}` — "
1700
+ "run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
1656
1701
  )
1657
- continue
1658
- positions.append(content.index(heading))
1659
- if len(positions) == len(_V2_AI_HANDOFF_HEADINGS) and positions != sorted(
1660
- positions
1661
- ):
1662
- failures.append(
1663
- "schema-v2 AI handoff markdown heading order does not match "
1664
- "templates/reports/final-report-v2.template.md."
1665
- )
1666
- for match in _V2_HUMAN_ONLY_RENDERED_RE.finditer(content):
1667
- field = match.group("heading") or match.group("label")
1668
- failures.append(
1669
- "schema-v2 AI handoff markdown contains human-only field "
1670
- f"{field!r}; render it only in the task-specific HTML."
1671
- )
1672
- if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1673
- failures.append(
1674
- "final report contains a `## 0. Reading Confirmation` heading — "
1675
- "Reading Confirmation lives in the worker audit sidecar, never "
1676
- "in the AI handoff markdown."
1677
- )
1702
+
1678
1703
 
1679
1704
  # Top-of-report Index block. The renderer
1680
1705
  # (scripts/okstra_ctl/render_final_report.py) injects `<a id="report-index">`
@@ -1782,16 +1807,34 @@ def _load_conformance_results(qa_dir: Path, manifest: dict) -> dict:
1782
1807
  return results
1783
1808
 
1784
1809
 
1785
- _DIFF_SUMMARY_RE = re.compile(r"^###\s+5\.7\.3\b.*?(?=^###\s|\Z)", re.MULTILINE | re.DOTALL)
1786
- _DIFF_ROW_PATH_RE = re.compile(r"^\|\s*`([^`]+)`\s*\|", re.MULTILINE)
1810
+ def _diff_summary_files(report_data: Mapping[str, Any]) -> list[str]:
1811
+ """implementation 리포트가 신고한 변경 파일 목록 (`implementation.diffSummary.files[].file`).
1787
1812
 
1813
+ 렌더된 §5.7.3 표를 정규식으로 긁던 자리다. 표는 data.json 의 이 배열에서
1814
+ 렌더되는 파생물이라, 표를 읽는 쪽은 렌더 형식이 바뀔 때마다 조용히 빈
1815
+ 목록을 돌려주고 — conformance / self-mock 두 게이트가 전부 통과로 열렸다.
1816
+ 스키마가 `implementation` 블록에서 `diffSummary` 를 required 로 잡고
1817
+ `rawStat` 이 비어있지 않으면 `files` 최소 1행을 요구하므로, 여기서는
1818
+ 구조가 어긋난 경우만 빈 목록으로 떨어뜨린다.
1788
1819
 
1789
- def _parse_diff_summary_files(content: str) -> list[str]:
1790
- """implementation 리포트 §5.7.3 Diff Summary 표의 첫 셀(백틱 경로)들을 추출."""
1791
- section = _DIFF_SUMMARY_RE.search(content)
1792
- if section is None:
1820
+ `diffSummary` 를 가진 task-type 은 implementation 뿐이다. final-verification
1821
+ 은 diff 를 `diffSummaryQuote` 문자열로만 인용하므로 두 게이트는 거기서
1822
+ (md 를 읽던 시절과 똑같이) vacuous 하다.
1823
+ """
1824
+ implementation = report_data.get("implementation")
1825
+ if not isinstance(implementation, Mapping):
1826
+ return []
1827
+ diff_summary = implementation.get("diffSummary")
1828
+ if not isinstance(diff_summary, Mapping):
1793
1829
  return []
1794
- return _DIFF_ROW_PATH_RE.findall(section.group(0))
1830
+ rows = diff_summary.get("files")
1831
+ if not isinstance(rows, list):
1832
+ return []
1833
+ return [
1834
+ row["file"]
1835
+ for row in rows
1836
+ if isinstance(row, Mapping) and isinstance(row.get("file"), str) and row["file"]
1837
+ ]
1795
1838
 
1796
1839
 
1797
1840
  _STAGE_RUN_DIR_RE = re.compile(r"^stage-\d+$")
@@ -2155,12 +2198,12 @@ def _validate_planning_conformance_declared(report_path: Path, failures: list[st
2155
2198
 
2156
2199
 
2157
2200
  def _validate_conformance_surfaces(
2158
- report_path: Path,
2201
+ report_data: Mapping[str, Any],
2159
2202
  scoped_manifest: dict,
2160
2203
  surface_patterns: object,
2161
2204
  failures: list[str],
2162
2205
  ) -> None:
2163
- changed_files = _parse_diff_summary_files(report_path.read_text(encoding="utf-8"))
2206
+ changed_files = _diff_summary_files(report_data)
2164
2207
  if not changed_files:
2165
2208
  return
2166
2209
  uncovered = (
@@ -2196,6 +2239,7 @@ def _validate_conformance(
2196
2239
  own stage suffix. Whole-task runs evaluate every entry.
2197
2240
  """
2198
2241
  warnings: list[str] = []
2242
+ report_data = _load_final_report_data(report_path)
2199
2243
  # conformance 산출물은 task-level(<task_root>/qa)에 있어 planning/
2200
2244
  # implementation/final-verification 가 공유한다. report_path 는
2201
2245
  # task_root/runs/<task-type>/reports/final-report.md (implementation 은
@@ -2236,7 +2280,7 @@ def _validate_conformance(
2236
2280
  f"{manifest_path} is absent"
2237
2281
  )
2238
2282
  _validate_conformance_surfaces(
2239
- report_path,
2283
+ report_data,
2240
2284
  empty_scoped_manifest,
2241
2285
  surface_patterns,
2242
2286
  failures,
@@ -2300,7 +2344,7 @@ def _validate_conformance(
2300
2344
  f"docs/superpowers/specs/2026-06-07-stage-conformance-qa-design.md."
2301
2345
  )
2302
2346
  _validate_conformance_surfaces(
2303
- report_path,
2347
+ report_data,
2304
2348
  scoped,
2305
2349
  surface_patterns,
2306
2350
  failures,
@@ -2638,13 +2682,13 @@ def _validate_selfmock(report_path: Path, failures: list[str]) -> None:
2638
2682
  Stage-isolated runs read their own `self-mock-stage-<N>.json`; whole-task runs
2639
2683
  read the flat `self-mock.json`.
2640
2684
 
2641
- Only the implementation template renders the §5.7.3 diff summary
2642
- (`templates/reports/final-report-v2.template.md:589`); a final-verification report
2643
- quotes the diff as a blockquote instead, so this gate is vacuous there by
2644
- design — self-mock is enforced at the implementation stage, and
2645
- final-verification is a read-only re-verify that adds no test files.
2685
+ Only an implementation report carries `implementation.diffSummary.files[]`;
2686
+ a final-verification report records the diff as the `diffSummaryQuote`
2687
+ string instead, so this gate is vacuous there by design — self-mock is
2688
+ enforced at the implementation stage, and final-verification is a read-only
2689
+ re-verify that adds no test files.
2646
2690
  """
2647
- changed = _parse_diff_summary_files(report_path.read_text(encoding="utf-8"))
2691
+ changed = _diff_summary_files(_load_final_report_data(report_path))
2648
2692
  test_files = [
2649
2693
  path
2650
2694
  for path in changed
@@ -2745,8 +2789,8 @@ def _check_selfmock_changed_files(
2745
2789
  `--changed-file` is what the mutation adapters select their production
2746
2790
  sources from. Omit it and every adapter gets an empty target set, which is
2747
2791
  not a finding but reads like one had been looked for. Requiring the sidecar
2748
- to account for every file in the §5.7.3 diff summary keeps "gate B saw this
2749
- stage" apart from "gate B was handed nothing".
2792
+ to account for every file in `implementation.diffSummary.files[]` keeps
2793
+ "gate B saw this stage" apart from "gate B was handed nothing".
2750
2794
 
2751
2795
  Coverage is asked of the WHOLE diff, not just its test files: the production
2752
2796
  sources are precisely the part gate A never looks at.
@@ -2771,7 +2815,8 @@ def _check_selfmock_changed_files(
2771
2815
  f"the mutation gate — {sidecar} records changedFiles={declared}. Each "
2772
2816
  "adapter picks its production sources out of that set, so a file left "
2773
2817
  "out is a file no mutant was ever generated for. Pass every path from "
2774
- "the §5.7.3 diff summary with `--changed-file`, production sources "
2818
+ "the report's `implementation.diffSummary.files[]` with `--changed-file`, "
2819
+ "production sources "
2775
2820
  "included — filtering to the test files leaves gate B nothing to run "
2776
2821
  "on and reports a pass it never earned."
2777
2822
  )
@@ -2835,6 +2880,10 @@ def validate_report(
2835
2880
  failures.append(f"final report is missing: {report_path}")
2836
2881
  return
2837
2882
 
2883
+ if (report_data or {}).get("schemaVersion") == "2.0":
2884
+ _validate_v2_report(report_data or {}, required_agent_status_entries, failures)
2885
+ return
2886
+
2838
2887
  content = report_path.read_text()
2839
2888
  for label in required_agent_status_entries:
2840
2889
  if label not in content:
@@ -2849,10 +2898,6 @@ def validate_report(
2849
2898
  "run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
2850
2899
  )
2851
2900
 
2852
- if (report_data or {}).get("schemaVersion") == "2.0":
2853
- _validate_v2_ai_handoff(content, failures)
2854
- return
2855
-
2856
2901
  # Catch the "workers typed `0` / `pending` instead of the placeholder"
2857
2902
  # failure mode that bypasses the placeholder check above.
2858
2903
  _scan_token_usage_summary(
@@ -3454,6 +3499,7 @@ def validate_final_report_data(
3454
3499
  )
3455
3500
  elif task_type == "implementation":
3456
3501
  _validate_stage_carry_sidecar_exists(data, report_path, failures)
3502
+ _validate_lead_authored_report(data, report_path, failures)
3457
3503
  if task_type == "error-analysis":
3458
3504
  _validate_error_analysis_consistency(data, failures)
3459
3505
  elif task_type == "final-verification":
@@ -6324,6 +6370,138 @@ def _validate_verifier_fail_blocks_verdict(data: dict, failures: list[str]) -> N
6324
6370
  )
6325
6371
 
6326
6372
 
6373
+
6374
+ _LEAD_AUTHORED = "Okstra lead"
6375
+ _REPORT_AUTHORING_HEADING_RE = re.compile(r"^## REPORT AUTHORING\s*$", re.MULTILINE)
6376
+ _REPORT_AUTHORING_APPROVED = "approved"
6377
+ # `report-writer.md` "Lead-authored fallback": the attempt must have reached one
6378
+ # of these with a concrete reason. `completed` means the worker produced the
6379
+ # report, so the lead had nothing to fall back from.
6380
+ _DISPATCH_FAILURE_STATUSES = {"error", "timeout", "not-run"}
6381
+
6382
+
6383
+ def _report_authoring_approval(report_path: Path) -> str:
6384
+ """The user's recorded answer on letting the lead author this report.
6385
+
6386
+ Read from the run's `user-responses/` sidecars, the same channel the
6387
+ clarification and plan-decision answers already use. The file is written by
6388
+ the user through `okstra user-response write`, which is the point: an
6389
+ approval the lead could author itself would be the self-report this gate
6390
+ exists to remove.
6391
+ """
6392
+ sidecar_dir = report_path.parent.parent / "user-responses"
6393
+ if not sidecar_dir.is_dir():
6394
+ return ""
6395
+ for sidecar in sorted(sidecar_dir.glob("*.md")):
6396
+ try:
6397
+ text = sidecar.read_text(encoding="utf-8")
6398
+ except OSError:
6399
+ continue
6400
+ match = _REPORT_AUTHORING_HEADING_RE.search(text)
6401
+ if not match:
6402
+ continue
6403
+ block = text[match.end():]
6404
+ next_heading = re.search(r"^## ", block, re.MULTILINE)
6405
+ if next_heading:
6406
+ block = block[: next_heading.start()]
6407
+ status = re.search(r"^-\s*Status:\s*(.+?)\s*$", block, re.MULTILINE)
6408
+ if status:
6409
+ return status.group(1).strip()
6410
+ return ""
6411
+
6412
+
6413
+ def _validate_lead_authored_report(
6414
+ data: dict,
6415
+ report_path: Path,
6416
+ failures: list[str],
6417
+ ) -> None:
6418
+ """A lead-authored final report needs a failed dispatch AND a user approval.
6419
+
6420
+ `header.reportAuthor` renders in the report but nothing read it, so a lead
6421
+ could name itself the author with no dispatch behind it. The contract has
6422
+ always required a real attempt that recorded a terminal failure with a
6423
+ reason; this adds the second door, because a lead that dispatches once,
6424
+ lets it fail, and proceeds has still decided alone. Neither door retires
6425
+ the other: an approval does not excuse a missing attempt, and an attempt
6426
+ that failed is the cue to ask, not the permission.
6427
+
6428
+ The record is not consumed by passing. The failure reason and the approval
6429
+ both stay on disk, and `header.reportAuthor` stays `Okstra lead` in the
6430
+ rendered report, so a later reader sees that this run took the fallback and
6431
+ why.
6432
+ """
6433
+ header = data.get("header")
6434
+ if not isinstance(header, Mapping):
6435
+ return
6436
+ if str(header.get("reportAuthor") or "").strip() != _LEAD_AUTHORED:
6437
+ return
6438
+ # release-handoff has no worker roster at all: it is single-lead by design,
6439
+ # so there is no dispatch to fail and nothing for the user to permit.
6440
+ if str(header.get("taskType") or "").strip() == "release-handoff":
6441
+ return
6442
+
6443
+ team_state = data.get("teamState")
6444
+ dispatches = (
6445
+ team_state.get("workerDispatches") if isinstance(team_state, Mapping) else None
6446
+ )
6447
+ attempts = [
6448
+ row
6449
+ for row in (dispatches if isinstance(dispatches, list) else [])
6450
+ if isinstance(row, Mapping)
6451
+ and str(row.get("workerId") or "").strip() == "report-writer"
6452
+ ]
6453
+ failed = [
6454
+ row
6455
+ for row in attempts
6456
+ if str(row.get("status") or "").strip() in _DISPATCH_FAILURE_STATUSES
6457
+ ]
6458
+ if not failed:
6459
+ failures.append(
6460
+ "final-report data.json: `header.reportAuthor` is `Okstra lead` but "
6461
+ "no report-writer dispatch recorded a terminal failure "
6462
+ f"({', '.join(sorted(_DISPATCH_FAILURE_STATUSES))}) in team-state. "
6463
+ "The lead-authored fallback is reachable only from an attempt that "
6464
+ "actually failed (prompts/lead/report-writer.md "
6465
+ "'Lead-authored fallback')"
6466
+ )
6467
+ elif not any(str(row.get("reason") or "").strip() for row in failed):
6468
+ failures.append(
6469
+ "final-report data.json: the report-writer dispatch failed but "
6470
+ "recorded no reason, so the lead-authored fallback rests on an "
6471
+ "unexplained failure. Record the tool error, the timeout, or the "
6472
+ "external blocker on the dispatch row"
6473
+ )
6474
+
6475
+ fallback = header.get("leadAuthoredFallback")
6476
+ if not isinstance(fallback, Mapping):
6477
+ failures.append(
6478
+ "final-report data.json: `header.reportAuthor` is `Okstra lead` but "
6479
+ "`header.leadAuthoredFallback` is absent. The approval passes the "
6480
+ "gate; it does not erase it — the failure reason and the approving "
6481
+ "sidecar belong in the report a human reads, not only in the "
6482
+ "sidecars they would have to go find"
6483
+ )
6484
+ else:
6485
+ recorded = str(fallback.get("dispatchFailureReason") or "").strip()
6486
+ reasons = {str(row.get("reason") or "").strip() for row in failed}
6487
+ if recorded and reasons and recorded not in reasons:
6488
+ failures.append(
6489
+ "final-report data.json: "
6490
+ "`header.leadAuthoredFallback.dispatchFailureReason` does not "
6491
+ "match any reason recorded on a failed report-writer dispatch. "
6492
+ "Quote the dispatch row verbatim rather than restating it"
6493
+ )
6494
+
6495
+ approval = _report_authoring_approval(report_path)
6496
+ if approval != _REPORT_AUTHORING_APPROVED:
6497
+ found = f"`{approval}`" if approval else "no `## REPORT AUTHORING` block"
6498
+ failures.append(
6499
+ "final-report data.json: `header.reportAuthor` is `Okstra lead` but "
6500
+ f"the run's `user-responses/` sidecars carry {found}. Only the user "
6501
+ "may permit the lead to author the report; ask at a gate and have "
6502
+ "the answer written through `okstra user-response write`"
6503
+ )
6504
+
6327
6505
  def _validate_stage_carry_sidecar_exists(
6328
6506
  data: dict,
6329
6507
  report_path: Path,
@@ -1034,7 +1034,7 @@ def _check_batch_cleanup_checkpoints(
1034
1034
  실제로 일어났는지 (prompts/profiles/_common-contract.md 'Phase-start cleanup').
1035
1035
  R1=convergence round 1 직전, R2=report-writer dispatch 직전(수렴이 있었으면
1036
1036
  마지막 라운드 이후 별도 1회). ISO-8601 ts 는 lexicographic 비교가 곧 시간순."""
1037
- detail = "prompts/profiles/_common-contract.md 'Phase-start cleanup'"
1037
+ detail = "prompts/lead/okstra-lead-contract.md 'Run-scoped worker-resource lifecycle'"
1038
1038
  cleanup_ts = sorted(ts for ts, _line in by_phase.get("phase-batch-cleanup", []))
1039
1039
  conv_ts = sorted(ts for ts, _line in by_phase.get("phase-5.5-convergence", []))
1040
1040
  synth_ts = sorted(ts for ts, _line in by_phase.get("phase-6-synthesis", []))