okstra 0.177.0 → 0.178.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/execute/team.mjs +14 -4
- package/dist/commands/execute/team.mjs.map +1 -1
- package/dist/commands/lifecycle/install.mjs +0 -1
- package/dist/commands/lifecycle/install.mjs.map +1 -1
- package/docs/architecture.md +3 -3
- package/docs/cli.md +1 -1
- package/docs/project-structure-overview.md +2 -2
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +1 -1
- package/runtime/bin/okstra-compact-reminder.sh +2 -2
- package/runtime/bin/okstra-render-report-views.py +13 -10
- package/runtime/prompts/lead/okstra-lead-contract.md +2 -2
- package/runtime/prompts/lead/report-writer.md +5 -1
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -6
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +23 -1
- package/runtime/python/okstra_ctl/agent_invocation.py +17 -0
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +13 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +96 -17
- package/runtime/python/okstra_ctl/dispatch_state.py +57 -0
- package/runtime/python/okstra_ctl/model_cli.py +11 -2
- package/runtime/python/okstra_ctl/model_discovery.py +12 -0
- package/runtime/python/okstra_ctl/pane_reclaim.py +49 -43
- package/runtime/python/okstra_ctl/render.py +53 -0
- package/runtime/python/okstra_ctl/report_html/render.py +7 -4
- package/runtime/python/okstra_ctl/report_views.py +35 -0
- package/runtime/python/okstra_ctl/run.py +88 -2
- package/runtime/python/okstra_ctl/team.py +84 -14
- package/runtime/python/okstra_ctl/tmux.py +2 -3
- package/runtime/python/okstra_ctl/user_response.py +20 -4
- package/runtime/python/okstra_ctl/worker_runner.py +2 -2
- package/runtime/python/okstra_ctl/write_policy.py +9 -1
- package/runtime/schemas/final-report-v2.0.schema.json +24 -1
- package/runtime/skills/okstra-run/SKILL.md +3 -3
- package/runtime/templates/reports/final-report-v2.template.md +2 -1
- package/runtime/templates/reports/html/base.template.html +1 -2
- package/runtime/validators/validate-report-views.py +30 -17
- package/runtime/validators/validate-run.py +256 -78
- package/runtime/validators/validate_session_conformance.py +1 -1
- package/runtime/bin/okstra-trace-cleanup.sh +0 -185
|
@@ -2,20 +2,26 @@
|
|
|
2
2
|
"""Validate the self-contained HTML view produced by Phase 7 step 1.5
|
|
3
3
|
(``scripts/okstra-render-report-views.py``).
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
5.
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
5
|
+
Takes the report's Markdown path as its handle and branches on schema.
|
|
6
|
+
|
|
7
|
+
A schema-v2 report is checked against its ``.data.json`` — the record the
|
|
8
|
+
HTML was rendered from — and never against the Markdown sibling, which is a
|
|
9
|
+
second rendering of that same record:
|
|
10
|
+
1. the ``*.html`` sibling exists;
|
|
11
|
+
2. the document carries the task type's own template;
|
|
12
|
+
3. every human field the task type requires is present, and no audit field
|
|
13
|
+
leaked into the human main;
|
|
14
|
+
4. each visualization's svg node ids match its fallback ids;
|
|
15
|
+
5. the Response IDs in the HTML match ``clarificationItems[]`` 1:1;
|
|
16
|
+
6. no external URL appears in ``<script src=>`` / ``<link href=>`` /
|
|
17
|
+
``<img src=>`` — the self-contained guarantee.
|
|
18
|
+
|
|
19
|
+
A schema-v1 report keeps the Markdown as its source, so its checks read the
|
|
20
|
+
§1 Clarification Items table (fail-closed when the heading is present but the
|
|
21
|
+
table will not parse), require the HTML sibling only when the report carries
|
|
22
|
+
clarification rows or an analysis-review / plan-approval contract, forbid form
|
|
23
|
+
controls in the §5.6 / §5.7 / §5.8 deliverable regions, and enforce the same
|
|
24
|
+
Response-ID parity and self-contained rules.
|
|
19
25
|
|
|
20
26
|
Exit codes: 0 on success, 1 on any failure. Failures are printed one
|
|
21
27
|
per line to stderr.
|
|
@@ -179,14 +185,21 @@ def _validate_analysis_review_html(
|
|
|
179
185
|
|
|
180
186
|
|
|
181
187
|
def validate(report_path: Path) -> list[str]:
|
|
188
|
+
"""Validate the html view of the report named by *report_path*.
|
|
189
|
+
|
|
190
|
+
A schema-v2 report is checked against its data.json and the rendered html;
|
|
191
|
+
the AI-handoff markdown is a sibling rendering of the same record and is
|
|
192
|
+
never read here, so it does not have to exist. Schema-v1 reports keep the
|
|
193
|
+
markdown as their source and still require it.
|
|
194
|
+
"""
|
|
182
195
|
failures: list[str] = []
|
|
196
|
+
v2 = _load_v2_data(report_path)
|
|
197
|
+
if v2 is not None:
|
|
198
|
+
return _validate_v2(report_path, v2[0], v2[1])
|
|
183
199
|
if not report_path.is_file():
|
|
184
200
|
return [f"final-report not found: {report_path}"]
|
|
185
201
|
|
|
186
202
|
md = report_path.read_text(encoding="utf-8")
|
|
187
|
-
v2 = _load_v2_data(report_path)
|
|
188
|
-
if v2 is not None:
|
|
189
|
-
return _validate_v2(report_path, v2[0], v2[1])
|
|
190
203
|
html_path = html_view_path(report_path)
|
|
191
204
|
# §1 헤딩이 있는데 파싱이 실패하면 md_ids 가 빈 []이 되어 "clarification 없음
|
|
192
205
|
# → skip" 으로 흘러 HTML form parity 게이트가 조용히 열린다. fail-closed.
|
|
@@ -51,6 +51,10 @@ from okstra_ctl.conformance import ( # noqa: E402
|
|
|
51
51
|
qa_result_from_dict,
|
|
52
52
|
validate_conformance_manifest,
|
|
53
53
|
)
|
|
54
|
+
from okstra_ctl.dispatch_state import ( # noqa: E402
|
|
55
|
+
DispatchError,
|
|
56
|
+
v2_worker_state_key,
|
|
57
|
+
)
|
|
54
58
|
from okstra_ctl.paths import RunRef # noqa: E402
|
|
55
59
|
from okstra_ctl.report_contract import CURRENT_REPORT_SCHEMA_VERSION # noqa: E402
|
|
56
60
|
from okstra_ctl.release_gate import ( # noqa: E402
|
|
@@ -862,6 +866,12 @@ def extract_contract(
|
|
|
862
866
|
required_worker_roles = []
|
|
863
867
|
failures.append("requiredWorkerRoles is missing from run/task manifest")
|
|
864
868
|
|
|
869
|
+
optional_worker_roles = run_contract.get("optionalWorkerRoles")
|
|
870
|
+
if not isinstance(optional_worker_roles, list):
|
|
871
|
+
optional_worker_roles = task_contract.get("optionalWorkerRoles")
|
|
872
|
+
if not isinstance(optional_worker_roles, list):
|
|
873
|
+
optional_worker_roles = []
|
|
874
|
+
|
|
865
875
|
lead_role = (
|
|
866
876
|
run_contract.get("leadRole")
|
|
867
877
|
or task_contract.get("leadRole")
|
|
@@ -892,6 +902,7 @@ def extract_contract(
|
|
|
892
902
|
or ""
|
|
893
903
|
),
|
|
894
904
|
"required_worker_roles": required_worker_roles,
|
|
905
|
+
"optional_worker_roles": optional_worker_roles,
|
|
895
906
|
"required_agent_status_entries": [
|
|
896
907
|
item
|
|
897
908
|
for item in required_agent_status_entries
|
|
@@ -1216,6 +1227,24 @@ def _is_legal_concurrent_run_skip(
|
|
|
1216
1227
|
)
|
|
1217
1228
|
|
|
1218
1229
|
|
|
1230
|
+
def _dispatch_roster_key(row: Mapping[str, Any]) -> str:
|
|
1231
|
+
"""The roster worker this dispatch row started, v1 or v2.
|
|
1232
|
+
|
|
1233
|
+
A v2 row carries no `workerId` — `_validate_agent_dispatch_contract`
|
|
1234
|
+
fails the run when one is present, calling it a v1/v2 identity mix. So
|
|
1235
|
+
reading the roster key off that field alone left every v2 row invisible
|
|
1236
|
+
and reported workers okstra had in fact started as never dispatched. The
|
|
1237
|
+
v2 projection is the one dispatch itself uses.
|
|
1238
|
+
"""
|
|
1239
|
+
worker_id = str(row.get("workerId", "")).strip()
|
|
1240
|
+
if worker_id:
|
|
1241
|
+
return worker_id
|
|
1242
|
+
try:
|
|
1243
|
+
return v2_worker_state_key(row)
|
|
1244
|
+
except DispatchError:
|
|
1245
|
+
return ""
|
|
1246
|
+
|
|
1247
|
+
|
|
1219
1248
|
def _validate_cmux_workers_were_dispatched_by_okstra(
|
|
1220
1249
|
team_state: dict,
|
|
1221
1250
|
workers: list,
|
|
@@ -1243,9 +1272,11 @@ def _validate_cmux_workers_were_dispatched_by_okstra(
|
|
|
1243
1272
|
if str(adapter.get("name", "")).strip() != "cmux":
|
|
1244
1273
|
return
|
|
1245
1274
|
recorded = {
|
|
1246
|
-
|
|
1275
|
+
key
|
|
1247
1276
|
for row in team_state.get("workerDispatches") or []
|
|
1248
1277
|
if isinstance(row, dict)
|
|
1278
|
+
for key in (_dispatch_roster_key(row),)
|
|
1279
|
+
if key
|
|
1249
1280
|
}
|
|
1250
1281
|
missing = []
|
|
1251
1282
|
for worker in workers:
|
|
@@ -1447,7 +1478,15 @@ def validate_team_state(
|
|
|
1447
1478
|
if status != "completed" and not reason:
|
|
1448
1479
|
failures.append(f"{role} with status `{status}` must include a reason")
|
|
1449
1480
|
|
|
1450
|
-
|
|
1481
|
+
# A declared optional role may appear in the roster and may equally be
|
|
1482
|
+
# absent: the run states which ones it can dispatch (today, the critics),
|
|
1483
|
+
# and running one is not a contract violation.
|
|
1484
|
+
optional_roles = {
|
|
1485
|
+
str(worker.get("role", "")).strip()
|
|
1486
|
+
for worker in contract.get("optional_worker_roles", [])
|
|
1487
|
+
if isinstance(worker, dict) and str(worker.get("role", "")).strip()
|
|
1488
|
+
}
|
|
1489
|
+
unexpected_roles = set(by_role) - set(expected_workers) - optional_roles
|
|
1451
1490
|
for role in sorted(unexpected_roles):
|
|
1452
1491
|
failures.append(f"unexpected worker role detected: {role}")
|
|
1453
1492
|
|
|
@@ -1621,60 +1660,46 @@ def _scan_token_usage_summary(
|
|
|
1621
1660
|
# a section heading line (not as inline text inside a paragraph or table).
|
|
1622
1661
|
_VERDICT_CARD_HEADING_RE = re.compile(r"^##[ \t]+Verdict Card\b", re.MULTILINE)
|
|
1623
1662
|
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
"
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
)
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1663
|
+
def _validate_v2_report(
|
|
1664
|
+
report_data: Mapping[str, Any],
|
|
1665
|
+
required_agent_status_entries: list[str],
|
|
1666
|
+
failures: list[str],
|
|
1667
|
+
) -> None:
|
|
1668
|
+
"""Contract checks for a schema-v2 report, read from its data.json.
|
|
1669
|
+
|
|
1670
|
+
The AI-handoff markdown is a deterministic rendering of this record, so the
|
|
1671
|
+
checks that used to scan it — heading count and order, human-only fields
|
|
1672
|
+
leaking into the AI artifact, a Reading Confirmation heading — were asking
|
|
1673
|
+
whether the renderer had done its job, not whether the report was sound.
|
|
1674
|
+
Schema enforcement in `render_final_report._enforce_schema` already answers
|
|
1675
|
+
the second question, and the first is settled by the template. What is left
|
|
1676
|
+
is the pair of facts the markdown could only ever carry second-hand: which
|
|
1677
|
+
agents the run must account for, and whether the token cells were filled.
|
|
1678
|
+
|
|
1679
|
+
Both are scanned over the serialized record, as text, because that is what
|
|
1680
|
+
the markdown scan was. Requiring each label to equal an
|
|
1681
|
+
`executionStatus[].role` would be a stricter rule than the contract states
|
|
1682
|
+
anywhere: the schema types `role` as a free string, no prompt or worker spec
|
|
1683
|
+
fixes its vocabulary, and `okstra_token_usage.report._match_worker_index`
|
|
1684
|
+
treats function-role spellings ("Analysis verifier", "Acceptance critic") as
|
|
1685
|
+
a shape reports do take — matching them by containment, never equality.
|
|
1686
|
+
Tightening this belongs with an authoring rule that says what to write, not
|
|
1687
|
+
on its own. Nothing is lost by reading the record instead of its rendering:
|
|
1688
|
+
the labels never come from the template or the i18n dictionaries.
|
|
1689
|
+
"""
|
|
1690
|
+
serialized = json.dumps(report_data, ensure_ascii=False)
|
|
1691
|
+
for label in required_agent_status_entries:
|
|
1692
|
+
if label not in serialized:
|
|
1653
1693
|
failures.append(
|
|
1654
|
-
"
|
|
1655
|
-
|
|
1694
|
+
f"final report does not include required agent status entry: {label}"
|
|
1695
|
+
)
|
|
1696
|
+
for placeholder in TOKEN_PLACEHOLDERS:
|
|
1697
|
+
if placeholder in serialized:
|
|
1698
|
+
failures.append(
|
|
1699
|
+
f"final report contains unsubstituted token placeholder `{placeholder}` — "
|
|
1700
|
+
"run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
|
|
1656
1701
|
)
|
|
1657
|
-
|
|
1658
|
-
positions.append(content.index(heading))
|
|
1659
|
-
if len(positions) == len(_V2_AI_HANDOFF_HEADINGS) and positions != sorted(
|
|
1660
|
-
positions
|
|
1661
|
-
):
|
|
1662
|
-
failures.append(
|
|
1663
|
-
"schema-v2 AI handoff markdown heading order does not match "
|
|
1664
|
-
"templates/reports/final-report-v2.template.md."
|
|
1665
|
-
)
|
|
1666
|
-
for match in _V2_HUMAN_ONLY_RENDERED_RE.finditer(content):
|
|
1667
|
-
field = match.group("heading") or match.group("label")
|
|
1668
|
-
failures.append(
|
|
1669
|
-
"schema-v2 AI handoff markdown contains human-only field "
|
|
1670
|
-
f"{field!r}; render it only in the task-specific HTML."
|
|
1671
|
-
)
|
|
1672
|
-
if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
|
|
1673
|
-
failures.append(
|
|
1674
|
-
"final report contains a `## 0. Reading Confirmation` heading — "
|
|
1675
|
-
"Reading Confirmation lives in the worker audit sidecar, never "
|
|
1676
|
-
"in the AI handoff markdown."
|
|
1677
|
-
)
|
|
1702
|
+
|
|
1678
1703
|
|
|
1679
1704
|
# Top-of-report Index block. The renderer
|
|
1680
1705
|
# (scripts/okstra_ctl/render_final_report.py) injects `<a id="report-index">`
|
|
@@ -1782,16 +1807,34 @@ def _load_conformance_results(qa_dir: Path, manifest: dict) -> dict:
|
|
|
1782
1807
|
return results
|
|
1783
1808
|
|
|
1784
1809
|
|
|
1785
|
-
|
|
1786
|
-
|
|
1810
|
+
def _diff_summary_files(report_data: Mapping[str, Any]) -> list[str]:
|
|
1811
|
+
"""implementation 리포트가 신고한 변경 파일 목록 (`implementation.diffSummary.files[].file`).
|
|
1787
1812
|
|
|
1813
|
+
렌더된 §5.7.3 표를 정규식으로 긁던 자리다. 표는 data.json 의 이 배열에서
|
|
1814
|
+
렌더되는 파생물이라, 표를 읽는 쪽은 렌더 형식이 바뀔 때마다 조용히 빈
|
|
1815
|
+
목록을 돌려주고 — conformance / self-mock 두 게이트가 전부 통과로 열렸다.
|
|
1816
|
+
스키마가 `implementation` 블록에서 `diffSummary` 를 required 로 잡고
|
|
1817
|
+
`rawStat` 이 비어있지 않으면 `files` 최소 1행을 요구하므로, 여기서는
|
|
1818
|
+
구조가 어긋난 경우만 빈 목록으로 떨어뜨린다.
|
|
1788
1819
|
|
|
1789
|
-
|
|
1790
|
-
|
|
1791
|
-
|
|
1792
|
-
|
|
1820
|
+
`diffSummary` 를 가진 task-type 은 implementation 뿐이다. final-verification
|
|
1821
|
+
은 diff 를 `diffSummaryQuote` 문자열로만 인용하므로 두 게이트는 거기서
|
|
1822
|
+
(md 를 읽던 시절과 똑같이) vacuous 하다.
|
|
1823
|
+
"""
|
|
1824
|
+
implementation = report_data.get("implementation")
|
|
1825
|
+
if not isinstance(implementation, Mapping):
|
|
1826
|
+
return []
|
|
1827
|
+
diff_summary = implementation.get("diffSummary")
|
|
1828
|
+
if not isinstance(diff_summary, Mapping):
|
|
1793
1829
|
return []
|
|
1794
|
-
|
|
1830
|
+
rows = diff_summary.get("files")
|
|
1831
|
+
if not isinstance(rows, list):
|
|
1832
|
+
return []
|
|
1833
|
+
return [
|
|
1834
|
+
row["file"]
|
|
1835
|
+
for row in rows
|
|
1836
|
+
if isinstance(row, Mapping) and isinstance(row.get("file"), str) and row["file"]
|
|
1837
|
+
]
|
|
1795
1838
|
|
|
1796
1839
|
|
|
1797
1840
|
_STAGE_RUN_DIR_RE = re.compile(r"^stage-\d+$")
|
|
@@ -2155,12 +2198,12 @@ def _validate_planning_conformance_declared(report_path: Path, failures: list[st
|
|
|
2155
2198
|
|
|
2156
2199
|
|
|
2157
2200
|
def _validate_conformance_surfaces(
|
|
2158
|
-
|
|
2201
|
+
report_data: Mapping[str, Any],
|
|
2159
2202
|
scoped_manifest: dict,
|
|
2160
2203
|
surface_patterns: object,
|
|
2161
2204
|
failures: list[str],
|
|
2162
2205
|
) -> None:
|
|
2163
|
-
changed_files =
|
|
2206
|
+
changed_files = _diff_summary_files(report_data)
|
|
2164
2207
|
if not changed_files:
|
|
2165
2208
|
return
|
|
2166
2209
|
uncovered = (
|
|
@@ -2196,6 +2239,7 @@ def _validate_conformance(
|
|
|
2196
2239
|
own stage suffix. Whole-task runs evaluate every entry.
|
|
2197
2240
|
"""
|
|
2198
2241
|
warnings: list[str] = []
|
|
2242
|
+
report_data = _load_final_report_data(report_path)
|
|
2199
2243
|
# conformance 산출물은 task-level(<task_root>/qa)에 있어 planning/
|
|
2200
2244
|
# implementation/final-verification 가 공유한다. report_path 는
|
|
2201
2245
|
# task_root/runs/<task-type>/reports/final-report.md (implementation 은
|
|
@@ -2236,7 +2280,7 @@ def _validate_conformance(
|
|
|
2236
2280
|
f"{manifest_path} is absent"
|
|
2237
2281
|
)
|
|
2238
2282
|
_validate_conformance_surfaces(
|
|
2239
|
-
|
|
2283
|
+
report_data,
|
|
2240
2284
|
empty_scoped_manifest,
|
|
2241
2285
|
surface_patterns,
|
|
2242
2286
|
failures,
|
|
@@ -2300,7 +2344,7 @@ def _validate_conformance(
|
|
|
2300
2344
|
f"docs/superpowers/specs/2026-06-07-stage-conformance-qa-design.md."
|
|
2301
2345
|
)
|
|
2302
2346
|
_validate_conformance_surfaces(
|
|
2303
|
-
|
|
2347
|
+
report_data,
|
|
2304
2348
|
scoped,
|
|
2305
2349
|
surface_patterns,
|
|
2306
2350
|
failures,
|
|
@@ -2638,13 +2682,13 @@ def _validate_selfmock(report_path: Path, failures: list[str]) -> None:
|
|
|
2638
2682
|
Stage-isolated runs read their own `self-mock-stage-<N>.json`; whole-task runs
|
|
2639
2683
|
read the flat `self-mock.json`.
|
|
2640
2684
|
|
|
2641
|
-
Only
|
|
2642
|
-
|
|
2643
|
-
|
|
2644
|
-
|
|
2645
|
-
|
|
2685
|
+
Only an implementation report carries `implementation.diffSummary.files[]`;
|
|
2686
|
+
a final-verification report records the diff as the `diffSummaryQuote`
|
|
2687
|
+
string instead, so this gate is vacuous there by design — self-mock is
|
|
2688
|
+
enforced at the implementation stage, and final-verification is a read-only
|
|
2689
|
+
re-verify that adds no test files.
|
|
2646
2690
|
"""
|
|
2647
|
-
changed =
|
|
2691
|
+
changed = _diff_summary_files(_load_final_report_data(report_path))
|
|
2648
2692
|
test_files = [
|
|
2649
2693
|
path
|
|
2650
2694
|
for path in changed
|
|
@@ -2745,8 +2789,8 @@ def _check_selfmock_changed_files(
|
|
|
2745
2789
|
`--changed-file` is what the mutation adapters select their production
|
|
2746
2790
|
sources from. Omit it and every adapter gets an empty target set, which is
|
|
2747
2791
|
not a finding but reads like one had been looked for. Requiring the sidecar
|
|
2748
|
-
to account for every file in
|
|
2749
|
-
stage" apart from "gate B was handed nothing".
|
|
2792
|
+
to account for every file in `implementation.diffSummary.files[]` keeps
|
|
2793
|
+
"gate B saw this stage" apart from "gate B was handed nothing".
|
|
2750
2794
|
|
|
2751
2795
|
Coverage is asked of the WHOLE diff, not just its test files: the production
|
|
2752
2796
|
sources are precisely the part gate A never looks at.
|
|
@@ -2771,7 +2815,8 @@ def _check_selfmock_changed_files(
|
|
|
2771
2815
|
f"the mutation gate — {sidecar} records changedFiles={declared}. Each "
|
|
2772
2816
|
"adapter picks its production sources out of that set, so a file left "
|
|
2773
2817
|
"out is a file no mutant was ever generated for. Pass every path from "
|
|
2774
|
-
"the
|
|
2818
|
+
"the report's `implementation.diffSummary.files[]` with `--changed-file`, "
|
|
2819
|
+
"production sources "
|
|
2775
2820
|
"included — filtering to the test files leaves gate B nothing to run "
|
|
2776
2821
|
"on and reports a pass it never earned."
|
|
2777
2822
|
)
|
|
@@ -2835,6 +2880,10 @@ def validate_report(
|
|
|
2835
2880
|
failures.append(f"final report is missing: {report_path}")
|
|
2836
2881
|
return
|
|
2837
2882
|
|
|
2883
|
+
if (report_data or {}).get("schemaVersion") == "2.0":
|
|
2884
|
+
_validate_v2_report(report_data or {}, required_agent_status_entries, failures)
|
|
2885
|
+
return
|
|
2886
|
+
|
|
2838
2887
|
content = report_path.read_text()
|
|
2839
2888
|
for label in required_agent_status_entries:
|
|
2840
2889
|
if label not in content:
|
|
@@ -2849,10 +2898,6 @@ def validate_report(
|
|
|
2849
2898
|
"run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
|
|
2850
2899
|
)
|
|
2851
2900
|
|
|
2852
|
-
if (report_data or {}).get("schemaVersion") == "2.0":
|
|
2853
|
-
_validate_v2_ai_handoff(content, failures)
|
|
2854
|
-
return
|
|
2855
|
-
|
|
2856
2901
|
# Catch the "workers typed `0` / `pending` instead of the placeholder"
|
|
2857
2902
|
# failure mode that bypasses the placeholder check above.
|
|
2858
2903
|
_scan_token_usage_summary(
|
|
@@ -3454,6 +3499,7 @@ def validate_final_report_data(
|
|
|
3454
3499
|
)
|
|
3455
3500
|
elif task_type == "implementation":
|
|
3456
3501
|
_validate_stage_carry_sidecar_exists(data, report_path, failures)
|
|
3502
|
+
_validate_lead_authored_report(data, report_path, failures)
|
|
3457
3503
|
if task_type == "error-analysis":
|
|
3458
3504
|
_validate_error_analysis_consistency(data, failures)
|
|
3459
3505
|
elif task_type == "final-verification":
|
|
@@ -6324,6 +6370,138 @@ def _validate_verifier_fail_blocks_verdict(data: dict, failures: list[str]) -> N
|
|
|
6324
6370
|
)
|
|
6325
6371
|
|
|
6326
6372
|
|
|
6373
|
+
|
|
6374
|
+
_LEAD_AUTHORED = "Okstra lead"
|
|
6375
|
+
_REPORT_AUTHORING_HEADING_RE = re.compile(r"^## REPORT AUTHORING\s*$", re.MULTILINE)
|
|
6376
|
+
_REPORT_AUTHORING_APPROVED = "approved"
|
|
6377
|
+
# `report-writer.md` "Lead-authored fallback": the attempt must have reached one
|
|
6378
|
+
# of these with a concrete reason. `completed` means the worker produced the
|
|
6379
|
+
# report, so the lead had nothing to fall back from.
|
|
6380
|
+
_DISPATCH_FAILURE_STATUSES = {"error", "timeout", "not-run"}
|
|
6381
|
+
|
|
6382
|
+
|
|
6383
|
+
def _report_authoring_approval(report_path: Path) -> str:
|
|
6384
|
+
"""The user's recorded answer on letting the lead author this report.
|
|
6385
|
+
|
|
6386
|
+
Read from the run's `user-responses/` sidecars, the same channel the
|
|
6387
|
+
clarification and plan-decision answers already use. The file is written by
|
|
6388
|
+
the user through `okstra user-response write`, which is the point: an
|
|
6389
|
+
approval the lead could author itself would be the self-report this gate
|
|
6390
|
+
exists to remove.
|
|
6391
|
+
"""
|
|
6392
|
+
sidecar_dir = report_path.parent.parent / "user-responses"
|
|
6393
|
+
if not sidecar_dir.is_dir():
|
|
6394
|
+
return ""
|
|
6395
|
+
for sidecar in sorted(sidecar_dir.glob("*.md")):
|
|
6396
|
+
try:
|
|
6397
|
+
text = sidecar.read_text(encoding="utf-8")
|
|
6398
|
+
except OSError:
|
|
6399
|
+
continue
|
|
6400
|
+
match = _REPORT_AUTHORING_HEADING_RE.search(text)
|
|
6401
|
+
if not match:
|
|
6402
|
+
continue
|
|
6403
|
+
block = text[match.end():]
|
|
6404
|
+
next_heading = re.search(r"^## ", block, re.MULTILINE)
|
|
6405
|
+
if next_heading:
|
|
6406
|
+
block = block[: next_heading.start()]
|
|
6407
|
+
status = re.search(r"^-\s*Status:\s*(.+?)\s*$", block, re.MULTILINE)
|
|
6408
|
+
if status:
|
|
6409
|
+
return status.group(1).strip()
|
|
6410
|
+
return ""
|
|
6411
|
+
|
|
6412
|
+
|
|
6413
|
+
def _validate_lead_authored_report(
|
|
6414
|
+
data: dict,
|
|
6415
|
+
report_path: Path,
|
|
6416
|
+
failures: list[str],
|
|
6417
|
+
) -> None:
|
|
6418
|
+
"""A lead-authored final report needs a failed dispatch AND a user approval.
|
|
6419
|
+
|
|
6420
|
+
`header.reportAuthor` renders in the report but nothing read it, so a lead
|
|
6421
|
+
could name itself the author with no dispatch behind it. The contract has
|
|
6422
|
+
always required a real attempt that recorded a terminal failure with a
|
|
6423
|
+
reason; this adds the second door, because a lead that dispatches once,
|
|
6424
|
+
lets it fail, and proceeds has still decided alone. Neither door retires
|
|
6425
|
+
the other: an approval does not excuse a missing attempt, and an attempt
|
|
6426
|
+
that failed is the cue to ask, not the permission.
|
|
6427
|
+
|
|
6428
|
+
The record is not consumed by passing. The failure reason and the approval
|
|
6429
|
+
both stay on disk, and `header.reportAuthor` stays `Okstra lead` in the
|
|
6430
|
+
rendered report, so a later reader sees that this run took the fallback and
|
|
6431
|
+
why.
|
|
6432
|
+
"""
|
|
6433
|
+
header = data.get("header")
|
|
6434
|
+
if not isinstance(header, Mapping):
|
|
6435
|
+
return
|
|
6436
|
+
if str(header.get("reportAuthor") or "").strip() != _LEAD_AUTHORED:
|
|
6437
|
+
return
|
|
6438
|
+
# release-handoff has no worker roster at all: it is single-lead by design,
|
|
6439
|
+
# so there is no dispatch to fail and nothing for the user to permit.
|
|
6440
|
+
if str(header.get("taskType") or "").strip() == "release-handoff":
|
|
6441
|
+
return
|
|
6442
|
+
|
|
6443
|
+
team_state = data.get("teamState")
|
|
6444
|
+
dispatches = (
|
|
6445
|
+
team_state.get("workerDispatches") if isinstance(team_state, Mapping) else None
|
|
6446
|
+
)
|
|
6447
|
+
attempts = [
|
|
6448
|
+
row
|
|
6449
|
+
for row in (dispatches if isinstance(dispatches, list) else [])
|
|
6450
|
+
if isinstance(row, Mapping)
|
|
6451
|
+
and str(row.get("workerId") or "").strip() == "report-writer"
|
|
6452
|
+
]
|
|
6453
|
+
failed = [
|
|
6454
|
+
row
|
|
6455
|
+
for row in attempts
|
|
6456
|
+
if str(row.get("status") or "").strip() in _DISPATCH_FAILURE_STATUSES
|
|
6457
|
+
]
|
|
6458
|
+
if not failed:
|
|
6459
|
+
failures.append(
|
|
6460
|
+
"final-report data.json: `header.reportAuthor` is `Okstra lead` but "
|
|
6461
|
+
"no report-writer dispatch recorded a terminal failure "
|
|
6462
|
+
f"({', '.join(sorted(_DISPATCH_FAILURE_STATUSES))}) in team-state. "
|
|
6463
|
+
"The lead-authored fallback is reachable only from an attempt that "
|
|
6464
|
+
"actually failed (prompts/lead/report-writer.md "
|
|
6465
|
+
"'Lead-authored fallback')"
|
|
6466
|
+
)
|
|
6467
|
+
elif not any(str(row.get("reason") or "").strip() for row in failed):
|
|
6468
|
+
failures.append(
|
|
6469
|
+
"final-report data.json: the report-writer dispatch failed but "
|
|
6470
|
+
"recorded no reason, so the lead-authored fallback rests on an "
|
|
6471
|
+
"unexplained failure. Record the tool error, the timeout, or the "
|
|
6472
|
+
"external blocker on the dispatch row"
|
|
6473
|
+
)
|
|
6474
|
+
|
|
6475
|
+
fallback = header.get("leadAuthoredFallback")
|
|
6476
|
+
if not isinstance(fallback, Mapping):
|
|
6477
|
+
failures.append(
|
|
6478
|
+
"final-report data.json: `header.reportAuthor` is `Okstra lead` but "
|
|
6479
|
+
"`header.leadAuthoredFallback` is absent. The approval passes the "
|
|
6480
|
+
"gate; it does not erase it — the failure reason and the approving "
|
|
6481
|
+
"sidecar belong in the report a human reads, not only in the "
|
|
6482
|
+
"sidecars they would have to go find"
|
|
6483
|
+
)
|
|
6484
|
+
else:
|
|
6485
|
+
recorded = str(fallback.get("dispatchFailureReason") or "").strip()
|
|
6486
|
+
reasons = {str(row.get("reason") or "").strip() for row in failed}
|
|
6487
|
+
if recorded and reasons and recorded not in reasons:
|
|
6488
|
+
failures.append(
|
|
6489
|
+
"final-report data.json: "
|
|
6490
|
+
"`header.leadAuthoredFallback.dispatchFailureReason` does not "
|
|
6491
|
+
"match any reason recorded on a failed report-writer dispatch. "
|
|
6492
|
+
"Quote the dispatch row verbatim rather than restating it"
|
|
6493
|
+
)
|
|
6494
|
+
|
|
6495
|
+
approval = _report_authoring_approval(report_path)
|
|
6496
|
+
if approval != _REPORT_AUTHORING_APPROVED:
|
|
6497
|
+
found = f"`{approval}`" if approval else "no `## REPORT AUTHORING` block"
|
|
6498
|
+
failures.append(
|
|
6499
|
+
"final-report data.json: `header.reportAuthor` is `Okstra lead` but "
|
|
6500
|
+
f"the run's `user-responses/` sidecars carry {found}. Only the user "
|
|
6501
|
+
"may permit the lead to author the report; ask at a gate and have "
|
|
6502
|
+
"the answer written through `okstra user-response write`"
|
|
6503
|
+
)
|
|
6504
|
+
|
|
6327
6505
|
def _validate_stage_carry_sidecar_exists(
|
|
6328
6506
|
data: dict,
|
|
6329
6507
|
report_path: Path,
|
|
@@ -1034,7 +1034,7 @@ def _check_batch_cleanup_checkpoints(
|
|
|
1034
1034
|
실제로 일어났는지 (prompts/profiles/_common-contract.md 'Phase-start cleanup').
|
|
1035
1035
|
R1=convergence round 1 직전, R2=report-writer dispatch 직전(수렴이 있었으면
|
|
1036
1036
|
마지막 라운드 이후 별도 1회). ISO-8601 ts 는 lexicographic 비교가 곧 시간순."""
|
|
1037
|
-
detail = "prompts/
|
|
1037
|
+
detail = "prompts/lead/okstra-lead-contract.md 'Run-scoped worker-resource lifecycle'"
|
|
1038
1038
|
cleanup_ts = sorted(ts for ts, _line in by_phase.get("phase-batch-cleanup", []))
|
|
1039
1039
|
conv_ts = sorted(ts for ts, _line in by_phase.get("phase-5.5-convergence", []))
|
|
1040
1040
|
synth_ts = sorted(ts for ts, _line in by_phase.get("phase-6-synthesis", []))
|