okstra 0.176.1 → 0.177.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/execute/team.mjs +14 -4
- package/dist/commands/execute/team.mjs.map +1 -1
- package/dist/commands/lifecycle/install.mjs +0 -1
- package/dist/commands/lifecycle/install.mjs.map +1 -1
- package/docs/architecture.md +3 -3
- package/docs/cli.md +1 -1
- package/docs/project-structure-overview.md +2 -2
- package/docs/task-process/final-verification.md +5 -3
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +2 -2
- package/runtime/bin/okstra-compact-reminder.sh +2 -2
- package/runtime/bin/okstra-provider-exec.py +2 -7
- package/runtime/bin/okstra-render-report-views.py +13 -10
- package/runtime/prompts/coding-preflight/overview.md +2 -1
- package/runtime/prompts/lead/convergence.md +11 -6
- package/runtime/prompts/lead/okstra-lead-contract.md +3 -3
- package/runtime/prompts/lead/plan-body-verification.md +4 -2
- package/runtime/prompts/lead/report-writer.md +8 -4
- package/runtime/prompts/profiles/_common-contract.md +2 -2
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +11 -9
- package/runtime/prompts/profiles/improvement-discovery.md +1 -1
- package/runtime/prompts/profiles/release-handoff.md +7 -6
- package/runtime/prompts/wizard/prompts.ko.json +2 -2
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -6
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +23 -1
- package/runtime/python/okstra_ctl/agent_invocation.py +17 -0
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +15 -2
- package/runtime/python/okstra_ctl/dispatch_core.py +140 -17
- package/runtime/python/okstra_ctl/dispatch_state.py +60 -3
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +49 -3
- package/runtime/python/okstra_ctl/handoff.py +27 -14
- package/runtime/python/okstra_ctl/model_cli.py +11 -2
- package/runtime/python/okstra_ctl/model_discovery.py +12 -0
- package/runtime/python/okstra_ctl/pane_reclaim.py +49 -43
- package/runtime/python/okstra_ctl/release_gate.py +56 -0
- package/runtime/python/okstra_ctl/render.py +53 -0
- package/runtime/python/okstra_ctl/report_contract.py +1 -0
- package/runtime/python/okstra_ctl/report_finalize.py +54 -0
- package/runtime/python/okstra_ctl/report_html/render.py +7 -4
- package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +4 -1
- package/runtime/python/okstra_ctl/run.py +119 -18
- package/runtime/python/okstra_ctl/stage_targets.py +73 -1
- package/runtime/python/okstra_ctl/team.py +84 -14
- package/runtime/python/okstra_ctl/tmux.py +2 -3
- package/runtime/python/okstra_ctl/wizard.py +19 -10
- package/runtime/python/okstra_ctl/worker_liveness.py +3 -1
- package/runtime/python/okstra_ctl/worker_runner.py +2 -2
- package/runtime/python/okstra_ctl/wrapper_status.py +15 -0
- package/runtime/python/okstra_ctl/write_policy.py +9 -1
- package/runtime/schemas/final-report-v2.0.schema.json +58 -19
- package/runtime/skills/okstra-run/SKILL.md +3 -3
- package/runtime/templates/reports/html/base.template.html +1 -2
- package/runtime/templates/reports/html/i18n/en.json +4 -0
- package/runtime/templates/reports/html/i18n/ko.json +4 -0
- package/runtime/templates/reports/html/tasks/final-verification.template.html +7 -1
- package/runtime/validators/validate-report-views.py +30 -17
- package/runtime/validators/validate-run.py +221 -90
- package/runtime/validators/validate_analysis_report.py +2 -5
- package/runtime/validators/validate_session_conformance.py +1 -1
- package/runtime/bin/okstra-trace-cleanup.sh +0 -185
|
@@ -51,8 +51,17 @@ from okstra_ctl.conformance import ( # noqa: E402
|
|
|
51
51
|
qa_result_from_dict,
|
|
52
52
|
validate_conformance_manifest,
|
|
53
53
|
)
|
|
54
|
+
from okstra_ctl.dispatch_state import ( # noqa: E402
|
|
55
|
+
DispatchError,
|
|
56
|
+
v2_worker_state_key,
|
|
57
|
+
)
|
|
54
58
|
from okstra_ctl.paths import RunRef # noqa: E402
|
|
55
59
|
from okstra_ctl.report_contract import CURRENT_REPORT_SCHEMA_VERSION # noqa: E402
|
|
60
|
+
from okstra_ctl.release_gate import ( # noqa: E402
|
|
61
|
+
RELEASE_HANDOFF_TARGETS,
|
|
62
|
+
blocking_condition_ids,
|
|
63
|
+
release_handoff_allowed,
|
|
64
|
+
)
|
|
56
65
|
from okstra_ctl.domain.host import HostNotRegistered # noqa: E402
|
|
57
66
|
from okstra_ctl.registry.host_registry import default_host_registry # noqa: E402
|
|
58
67
|
from okstra_ctl.build_tools import ( # noqa: E402
|
|
@@ -857,6 +866,12 @@ def extract_contract(
|
|
|
857
866
|
required_worker_roles = []
|
|
858
867
|
failures.append("requiredWorkerRoles is missing from run/task manifest")
|
|
859
868
|
|
|
869
|
+
optional_worker_roles = run_contract.get("optionalWorkerRoles")
|
|
870
|
+
if not isinstance(optional_worker_roles, list):
|
|
871
|
+
optional_worker_roles = task_contract.get("optionalWorkerRoles")
|
|
872
|
+
if not isinstance(optional_worker_roles, list):
|
|
873
|
+
optional_worker_roles = []
|
|
874
|
+
|
|
860
875
|
lead_role = (
|
|
861
876
|
run_contract.get("leadRole")
|
|
862
877
|
or task_contract.get("leadRole")
|
|
@@ -887,6 +902,7 @@ def extract_contract(
|
|
|
887
902
|
or ""
|
|
888
903
|
),
|
|
889
904
|
"required_worker_roles": required_worker_roles,
|
|
905
|
+
"optional_worker_roles": optional_worker_roles,
|
|
890
906
|
"required_agent_status_entries": [
|
|
891
907
|
item
|
|
892
908
|
for item in required_agent_status_entries
|
|
@@ -1211,6 +1227,24 @@ def _is_legal_concurrent_run_skip(
|
|
|
1211
1227
|
)
|
|
1212
1228
|
|
|
1213
1229
|
|
|
1230
|
+
def _dispatch_roster_key(row: Mapping[str, Any]) -> str:
|
|
1231
|
+
"""The roster worker this dispatch row started, v1 or v2.
|
|
1232
|
+
|
|
1233
|
+
A v2 row carries no `workerId` — `_validate_agent_dispatch_contract`
|
|
1234
|
+
fails the run when one is present, calling it a v1/v2 identity mix. So
|
|
1235
|
+
reading the roster key off that field alone left every v2 row invisible
|
|
1236
|
+
and reported workers okstra had in fact started as never dispatched. The
|
|
1237
|
+
v2 projection is the one dispatch itself uses.
|
|
1238
|
+
"""
|
|
1239
|
+
worker_id = str(row.get("workerId", "")).strip()
|
|
1240
|
+
if worker_id:
|
|
1241
|
+
return worker_id
|
|
1242
|
+
try:
|
|
1243
|
+
return v2_worker_state_key(row)
|
|
1244
|
+
except DispatchError:
|
|
1245
|
+
return ""
|
|
1246
|
+
|
|
1247
|
+
|
|
1214
1248
|
def _validate_cmux_workers_were_dispatched_by_okstra(
|
|
1215
1249
|
team_state: dict,
|
|
1216
1250
|
workers: list,
|
|
@@ -1238,9 +1272,11 @@ def _validate_cmux_workers_were_dispatched_by_okstra(
|
|
|
1238
1272
|
if str(adapter.get("name", "")).strip() != "cmux":
|
|
1239
1273
|
return
|
|
1240
1274
|
recorded = {
|
|
1241
|
-
|
|
1275
|
+
key
|
|
1242
1276
|
for row in team_state.get("workerDispatches") or []
|
|
1243
1277
|
if isinstance(row, dict)
|
|
1278
|
+
for key in (_dispatch_roster_key(row),)
|
|
1279
|
+
if key
|
|
1244
1280
|
}
|
|
1245
1281
|
missing = []
|
|
1246
1282
|
for worker in workers:
|
|
@@ -1442,7 +1478,15 @@ def validate_team_state(
|
|
|
1442
1478
|
if status != "completed" and not reason:
|
|
1443
1479
|
failures.append(f"{role} with status `{status}` must include a reason")
|
|
1444
1480
|
|
|
1445
|
-
|
|
1481
|
+
# A declared optional role may appear in the roster and may equally be
|
|
1482
|
+
# absent: the run states which ones it can dispatch (today, the critics),
|
|
1483
|
+
# and running one is not a contract violation.
|
|
1484
|
+
optional_roles = {
|
|
1485
|
+
str(worker.get("role", "")).strip()
|
|
1486
|
+
for worker in contract.get("optional_worker_roles", [])
|
|
1487
|
+
if isinstance(worker, dict) and str(worker.get("role", "")).strip()
|
|
1488
|
+
}
|
|
1489
|
+
unexpected_roles = set(by_role) - set(expected_workers) - optional_roles
|
|
1446
1490
|
for role in sorted(unexpected_roles):
|
|
1447
1491
|
failures.append(f"unexpected worker role detected: {role}")
|
|
1448
1492
|
|
|
@@ -1616,60 +1660,46 @@ def _scan_token_usage_summary(
|
|
|
1616
1660
|
# a section heading line (not as inline text inside a paragraph or table).
|
|
1617
1661
|
_VERDICT_CARD_HEADING_RE = re.compile(r"^##[ \t]+Verdict Card\b", re.MULTILINE)
|
|
1618
1662
|
|
|
1619
|
-
|
|
1620
|
-
|
|
1621
|
-
|
|
1622
|
-
|
|
1623
|
-
|
|
1624
|
-
"
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
|
|
1630
|
-
|
|
1631
|
-
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
)
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1663
|
+
def _validate_v2_report(
|
|
1664
|
+
report_data: Mapping[str, Any],
|
|
1665
|
+
required_agent_status_entries: list[str],
|
|
1666
|
+
failures: list[str],
|
|
1667
|
+
) -> None:
|
|
1668
|
+
"""Contract checks for a schema-v2 report, read from its data.json.
|
|
1669
|
+
|
|
1670
|
+
The AI-handoff markdown is a deterministic rendering of this record, so the
|
|
1671
|
+
checks that used to scan it — heading count and order, human-only fields
|
|
1672
|
+
leaking into the AI artifact, a Reading Confirmation heading — were asking
|
|
1673
|
+
whether the renderer had done its job, not whether the report was sound.
|
|
1674
|
+
Schema enforcement in `render_final_report._enforce_schema` already answers
|
|
1675
|
+
the second question, and the first is settled by the template. What is left
|
|
1676
|
+
is the pair of facts the markdown could only ever carry second-hand: which
|
|
1677
|
+
agents the run must account for, and whether the token cells were filled.
|
|
1678
|
+
|
|
1679
|
+
Both are scanned over the serialized record, as text, because that is what
|
|
1680
|
+
the markdown scan was. Requiring each label to equal an
|
|
1681
|
+
`executionStatus[].role` would be a stricter rule than the contract states
|
|
1682
|
+
anywhere: the schema types `role` as a free string, no prompt or worker spec
|
|
1683
|
+
fixes its vocabulary, and `okstra_token_usage.report._match_worker_index`
|
|
1684
|
+
treats function-role spellings ("Analysis verifier", "Acceptance critic") as
|
|
1685
|
+
a shape reports do take — matching them by containment, never equality.
|
|
1686
|
+
Tightening this belongs with an authoring rule that says what to write, not
|
|
1687
|
+
on its own. Nothing is lost by reading the record instead of its rendering:
|
|
1688
|
+
the labels never come from the template or the i18n dictionaries.
|
|
1689
|
+
"""
|
|
1690
|
+
serialized = json.dumps(report_data, ensure_ascii=False)
|
|
1691
|
+
for label in required_agent_status_entries:
|
|
1692
|
+
if label not in serialized:
|
|
1648
1693
|
failures.append(
|
|
1649
|
-
"
|
|
1650
|
-
|
|
1694
|
+
f"final report does not include required agent status entry: {label}"
|
|
1695
|
+
)
|
|
1696
|
+
for placeholder in TOKEN_PLACEHOLDERS:
|
|
1697
|
+
if placeholder in serialized:
|
|
1698
|
+
failures.append(
|
|
1699
|
+
f"final report contains unsubstituted token placeholder `{placeholder}` — "
|
|
1700
|
+
"run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
|
|
1651
1701
|
)
|
|
1652
|
-
|
|
1653
|
-
positions.append(content.index(heading))
|
|
1654
|
-
if len(positions) == len(_V2_AI_HANDOFF_HEADINGS) and positions != sorted(
|
|
1655
|
-
positions
|
|
1656
|
-
):
|
|
1657
|
-
failures.append(
|
|
1658
|
-
"schema-v2 AI handoff markdown heading order does not match "
|
|
1659
|
-
"templates/reports/final-report-v2.template.md."
|
|
1660
|
-
)
|
|
1661
|
-
for match in _V2_HUMAN_ONLY_RENDERED_RE.finditer(content):
|
|
1662
|
-
field = match.group("heading") or match.group("label")
|
|
1663
|
-
failures.append(
|
|
1664
|
-
"schema-v2 AI handoff markdown contains human-only field "
|
|
1665
|
-
f"{field!r}; render it only in the task-specific HTML."
|
|
1666
|
-
)
|
|
1667
|
-
if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
|
|
1668
|
-
failures.append(
|
|
1669
|
-
"final report contains a `## 0. Reading Confirmation` heading — "
|
|
1670
|
-
"Reading Confirmation lives in the worker audit sidecar, never "
|
|
1671
|
-
"in the AI handoff markdown."
|
|
1672
|
-
)
|
|
1702
|
+
|
|
1673
1703
|
|
|
1674
1704
|
# Top-of-report Index block. The renderer
|
|
1675
1705
|
# (scripts/okstra_ctl/render_final_report.py) injects `<a id="report-index">`
|
|
@@ -1777,16 +1807,34 @@ def _load_conformance_results(qa_dir: Path, manifest: dict) -> dict:
|
|
|
1777
1807
|
return results
|
|
1778
1808
|
|
|
1779
1809
|
|
|
1780
|
-
|
|
1781
|
-
|
|
1810
|
+
def _diff_summary_files(report_data: Mapping[str, Any]) -> list[str]:
|
|
1811
|
+
"""implementation 리포트가 신고한 변경 파일 목록 (`implementation.diffSummary.files[].file`).
|
|
1782
1812
|
|
|
1813
|
+
렌더된 §5.7.3 표를 정규식으로 긁던 자리다. 표는 data.json 의 이 배열에서
|
|
1814
|
+
렌더되는 파생물이라, 표를 읽는 쪽은 렌더 형식이 바뀔 때마다 조용히 빈
|
|
1815
|
+
목록을 돌려주고 — conformance / self-mock 두 게이트가 전부 통과로 열렸다.
|
|
1816
|
+
스키마가 `implementation` 블록에서 `diffSummary` 를 required 로 잡고
|
|
1817
|
+
`rawStat` 이 비어있지 않으면 `files` 최소 1행을 요구하므로, 여기서는
|
|
1818
|
+
구조가 어긋난 경우만 빈 목록으로 떨어뜨린다.
|
|
1783
1819
|
|
|
1784
|
-
|
|
1785
|
-
|
|
1786
|
-
|
|
1787
|
-
|
|
1820
|
+
`diffSummary` 를 가진 task-type 은 implementation 뿐이다. final-verification
|
|
1821
|
+
은 diff 를 `diffSummaryQuote` 문자열로만 인용하므로 두 게이트는 거기서
|
|
1822
|
+
(md 를 읽던 시절과 똑같이) vacuous 하다.
|
|
1823
|
+
"""
|
|
1824
|
+
implementation = report_data.get("implementation")
|
|
1825
|
+
if not isinstance(implementation, Mapping):
|
|
1826
|
+
return []
|
|
1827
|
+
diff_summary = implementation.get("diffSummary")
|
|
1828
|
+
if not isinstance(diff_summary, Mapping):
|
|
1788
1829
|
return []
|
|
1789
|
-
|
|
1830
|
+
rows = diff_summary.get("files")
|
|
1831
|
+
if not isinstance(rows, list):
|
|
1832
|
+
return []
|
|
1833
|
+
return [
|
|
1834
|
+
row["file"]
|
|
1835
|
+
for row in rows
|
|
1836
|
+
if isinstance(row, Mapping) and isinstance(row.get("file"), str) and row["file"]
|
|
1837
|
+
]
|
|
1790
1838
|
|
|
1791
1839
|
|
|
1792
1840
|
_STAGE_RUN_DIR_RE = re.compile(r"^stage-\d+$")
|
|
@@ -2150,12 +2198,12 @@ def _validate_planning_conformance_declared(report_path: Path, failures: list[st
|
|
|
2150
2198
|
|
|
2151
2199
|
|
|
2152
2200
|
def _validate_conformance_surfaces(
|
|
2153
|
-
|
|
2201
|
+
report_data: Mapping[str, Any],
|
|
2154
2202
|
scoped_manifest: dict,
|
|
2155
2203
|
surface_patterns: object,
|
|
2156
2204
|
failures: list[str],
|
|
2157
2205
|
) -> None:
|
|
2158
|
-
changed_files =
|
|
2206
|
+
changed_files = _diff_summary_files(report_data)
|
|
2159
2207
|
if not changed_files:
|
|
2160
2208
|
return
|
|
2161
2209
|
uncovered = (
|
|
@@ -2191,6 +2239,7 @@ def _validate_conformance(
|
|
|
2191
2239
|
own stage suffix. Whole-task runs evaluate every entry.
|
|
2192
2240
|
"""
|
|
2193
2241
|
warnings: list[str] = []
|
|
2242
|
+
report_data = _load_final_report_data(report_path)
|
|
2194
2243
|
# conformance 산출물은 task-level(<task_root>/qa)에 있어 planning/
|
|
2195
2244
|
# implementation/final-verification 가 공유한다. report_path 는
|
|
2196
2245
|
# task_root/runs/<task-type>/reports/final-report.md (implementation 은
|
|
@@ -2231,7 +2280,7 @@ def _validate_conformance(
|
|
|
2231
2280
|
f"{manifest_path} is absent"
|
|
2232
2281
|
)
|
|
2233
2282
|
_validate_conformance_surfaces(
|
|
2234
|
-
|
|
2283
|
+
report_data,
|
|
2235
2284
|
empty_scoped_manifest,
|
|
2236
2285
|
surface_patterns,
|
|
2237
2286
|
failures,
|
|
@@ -2295,7 +2344,7 @@ def _validate_conformance(
|
|
|
2295
2344
|
f"docs/superpowers/specs/2026-06-07-stage-conformance-qa-design.md."
|
|
2296
2345
|
)
|
|
2297
2346
|
_validate_conformance_surfaces(
|
|
2298
|
-
|
|
2347
|
+
report_data,
|
|
2299
2348
|
scoped,
|
|
2300
2349
|
surface_patterns,
|
|
2301
2350
|
failures,
|
|
@@ -2633,13 +2682,13 @@ def _validate_selfmock(report_path: Path, failures: list[str]) -> None:
|
|
|
2633
2682
|
Stage-isolated runs read their own `self-mock-stage-<N>.json`; whole-task runs
|
|
2634
2683
|
read the flat `self-mock.json`.
|
|
2635
2684
|
|
|
2636
|
-
Only
|
|
2637
|
-
|
|
2638
|
-
|
|
2639
|
-
|
|
2640
|
-
|
|
2685
|
+
Only an implementation report carries `implementation.diffSummary.files[]`;
|
|
2686
|
+
a final-verification report records the diff as the `diffSummaryQuote`
|
|
2687
|
+
string instead, so this gate is vacuous there by design — self-mock is
|
|
2688
|
+
enforced at the implementation stage, and final-verification is a read-only
|
|
2689
|
+
re-verify that adds no test files.
|
|
2641
2690
|
"""
|
|
2642
|
-
changed =
|
|
2691
|
+
changed = _diff_summary_files(_load_final_report_data(report_path))
|
|
2643
2692
|
test_files = [
|
|
2644
2693
|
path
|
|
2645
2694
|
for path in changed
|
|
@@ -2740,8 +2789,8 @@ def _check_selfmock_changed_files(
|
|
|
2740
2789
|
`--changed-file` is what the mutation adapters select their production
|
|
2741
2790
|
sources from. Omit it and every adapter gets an empty target set, which is
|
|
2742
2791
|
not a finding but reads like one had been looked for. Requiring the sidecar
|
|
2743
|
-
to account for every file in
|
|
2744
|
-
stage" apart from "gate B was handed nothing".
|
|
2792
|
+
to account for every file in `implementation.diffSummary.files[]` keeps
|
|
2793
|
+
"gate B saw this stage" apart from "gate B was handed nothing".
|
|
2745
2794
|
|
|
2746
2795
|
Coverage is asked of the WHOLE diff, not just its test files: the production
|
|
2747
2796
|
sources are precisely the part gate A never looks at.
|
|
@@ -2766,7 +2815,8 @@ def _check_selfmock_changed_files(
|
|
|
2766
2815
|
f"the mutation gate — {sidecar} records changedFiles={declared}. Each "
|
|
2767
2816
|
"adapter picks its production sources out of that set, so a file left "
|
|
2768
2817
|
"out is a file no mutant was ever generated for. Pass every path from "
|
|
2769
|
-
"the
|
|
2818
|
+
"the report's `implementation.diffSummary.files[]` with `--changed-file`, "
|
|
2819
|
+
"production sources "
|
|
2770
2820
|
"included — filtering to the test files leaves gate B nothing to run "
|
|
2771
2821
|
"on and reports a pass it never earned."
|
|
2772
2822
|
)
|
|
@@ -2830,6 +2880,10 @@ def validate_report(
|
|
|
2830
2880
|
failures.append(f"final report is missing: {report_path}")
|
|
2831
2881
|
return
|
|
2832
2882
|
|
|
2883
|
+
if (report_data or {}).get("schemaVersion") == "2.0":
|
|
2884
|
+
_validate_v2_report(report_data or {}, required_agent_status_entries, failures)
|
|
2885
|
+
return
|
|
2886
|
+
|
|
2833
2887
|
content = report_path.read_text()
|
|
2834
2888
|
for label in required_agent_status_entries:
|
|
2835
2889
|
if label not in content:
|
|
@@ -2844,10 +2898,6 @@ def validate_report(
|
|
|
2844
2898
|
"run `okstra-token-usage.py ... --substitute-data <report-path>` during Phase 7"
|
|
2845
2899
|
)
|
|
2846
2900
|
|
|
2847
|
-
if (report_data or {}).get("schemaVersion") == "2.0":
|
|
2848
|
-
_validate_v2_ai_handoff(content, failures)
|
|
2849
|
-
return
|
|
2850
|
-
|
|
2851
2901
|
# Catch the "workers typed `0` / `pending` instead of the placeholder"
|
|
2852
2902
|
# failure mode that bypasses the placeholder check above.
|
|
2853
2903
|
_scan_token_usage_summary(
|
|
@@ -6247,7 +6297,7 @@ def _validate_verified_row_recorded(
|
|
|
6247
6297
|
report_path: Path,
|
|
6248
6298
|
failures: list[str],
|
|
6249
6299
|
) -> None:
|
|
6250
|
-
"""
|
|
6300
|
+
"""A release-ready single-stage verification must leave its `verified` row.
|
|
6251
6301
|
|
|
6252
6302
|
`okstra handoff record-verified` validates its own inputs, but nothing
|
|
6253
6303
|
checked that it ever ran. Skipping it leaves the report saying `accepted`
|
|
@@ -6258,8 +6308,7 @@ def _validate_verified_row_recorded(
|
|
|
6258
6308
|
"""
|
|
6259
6309
|
if str(data.get("verificationScope") or "") != "single-stage":
|
|
6260
6310
|
return
|
|
6261
|
-
|
|
6262
|
-
if token.lower() != "accepted":
|
|
6311
|
+
if not release_handoff_allowed(data):
|
|
6263
6312
|
return
|
|
6264
6313
|
stages = {
|
|
6265
6314
|
row.get("stage")
|
|
@@ -6279,7 +6328,7 @@ def _validate_verified_row_recorded(
|
|
|
6279
6328
|
missing = sorted(s for s in stages if s not in verified)
|
|
6280
6329
|
if missing:
|
|
6281
6330
|
failures.append(
|
|
6282
|
-
f"final-verification
|
|
6331
|
+
f"final-verification cleared stage(s) {missing} for release but "
|
|
6283
6332
|
"`runs/implementation-planning/consumers.jsonl` carries no "
|
|
6284
6333
|
"`verified` row for them. Run `okstra handoff record-verified` "
|
|
6285
6334
|
"before finishing — without that row the report says accepted "
|
|
@@ -6308,11 +6357,11 @@ def _validate_verifier_fail_blocks_verdict(data: dict, failures: list[str]) -> N
|
|
|
6308
6357
|
})
|
|
6309
6358
|
if not failed:
|
|
6310
6359
|
return
|
|
6311
|
-
token = str((data.get("
|
|
6360
|
+
token = str((data.get("finalVerdict") or {}).get("verdictToken") or "").strip()
|
|
6312
6361
|
if token in _PASSING_VERDICT_TOKENS:
|
|
6313
6362
|
failures.append(
|
|
6314
6363
|
f"final-report data.json: verifier(s) {failed} recorded "
|
|
6315
|
-
f"`verdict: FAIL` but `
|
|
6364
|
+
f"`verdict: FAIL` but `finalVerdict.verdictToken` is `{token}`. A "
|
|
6316
6365
|
"verifier rejection MUST survive into the published verdict — "
|
|
6317
6366
|
"dropping it during synthesis is how rejected work reaches "
|
|
6318
6367
|
"`release-handoff`. Either carry the FAIL into a blocking verdict "
|
|
@@ -7914,6 +7963,82 @@ def _validate_stage_has_requirement(data: dict, failures: list[str]) -> None:
|
|
|
7914
7963
|
)
|
|
7915
7964
|
|
|
7916
7965
|
|
|
7966
|
+
_ADDED_SURFACE_NO_CALLER_RE = re.compile(r"^\s*none\b", re.IGNORECASE)
|
|
7967
|
+
|
|
7968
|
+
|
|
7969
|
+
def _validate_added_surface_audit(data: dict, failures: list[str]) -> None:
|
|
7970
|
+
"""Every surface the diff added is traced to a requirement, exempted, or paid for.
|
|
7971
|
+
|
|
7972
|
+
The coverage table proves each requirement reached the diff. Nothing proved
|
|
7973
|
+
the reverse — that each thing the diff added answers a requirement — so work
|
|
7974
|
+
nobody asked for passed every gate. This check reads the reverse table and
|
|
7975
|
+
refuses a row that calls itself over-delivery without the blocker or
|
|
7976
|
+
condition it became: a caller-less surface is an acceptance blocker, and a
|
|
7977
|
+
surface with callers but no requirement is a conditional-acceptance
|
|
7978
|
+
condition (ADR-0009 grades the two differently on purpose).
|
|
7979
|
+
"""
|
|
7980
|
+
fv = data.get("finalVerification")
|
|
7981
|
+
if not isinstance(fv, Mapping):
|
|
7982
|
+
return
|
|
7983
|
+
rows = fv.get("addedSurfaceAudit")
|
|
7984
|
+
if not isinstance(rows, list):
|
|
7985
|
+
return
|
|
7986
|
+
blocker_ids = {
|
|
7987
|
+
str(row.get("id"))
|
|
7988
|
+
for row in (fv.get("acceptanceBlockers") or [])
|
|
7989
|
+
if isinstance(row, Mapping)
|
|
7990
|
+
}
|
|
7991
|
+
condition_ids = {
|
|
7992
|
+
str(row.get("id"))
|
|
7993
|
+
for row in ((data.get("finalVerdict") or {}).get(
|
|
7994
|
+
"conditionalAcceptanceConditions") or [])
|
|
7995
|
+
if isinstance(row, Mapping)
|
|
7996
|
+
}
|
|
7997
|
+
for row in rows:
|
|
7998
|
+
if not isinstance(row, Mapping):
|
|
7999
|
+
continue
|
|
8000
|
+
row_id = str(row.get("id") or "<id 없음>")
|
|
8001
|
+
disposition = str(row.get("disposition") or "")
|
|
8002
|
+
note = str(row.get("note") or "")
|
|
8003
|
+
if disposition == "traced" and not str(row.get("requirement") or "").strip():
|
|
8004
|
+
failures.append(
|
|
8005
|
+
f"final-verification: addedSurfaceAudit {row_id} is `traced` but "
|
|
8006
|
+
"names no requirement — a surface is traced to something the "
|
|
8007
|
+
"brief asked for, or it is not traced."
|
|
8008
|
+
)
|
|
8009
|
+
continue
|
|
8010
|
+
if disposition != "over-delivery":
|
|
8011
|
+
continue
|
|
8012
|
+
caller_less = bool(
|
|
8013
|
+
_ADDED_SURFACE_NO_CALLER_RE.match(str(row.get("callers") or ""))
|
|
8014
|
+
)
|
|
8015
|
+
expected, known = (
|
|
8016
|
+
("AB", blocker_ids) if caller_less else ("CA", condition_ids)
|
|
8017
|
+
)
|
|
8018
|
+
cited = set(re.findall(rf"\b{expected}-\d{{3,}}\b", note))
|
|
8019
|
+
if not cited:
|
|
8020
|
+
failures.append(
|
|
8021
|
+
f"final-verification: addedSurfaceAudit {row_id} is "
|
|
8022
|
+
f"`over-delivery` with callers "
|
|
8023
|
+
f"{'none' if caller_less else 'recorded'}, so its note MUST cite "
|
|
8024
|
+
f"the `{expected}-NNN` row it became — "
|
|
8025
|
+
+ (
|
|
8026
|
+
"a caller-less surface is an acceptance blocker"
|
|
8027
|
+
if caller_less
|
|
8028
|
+
else "a surface with callers but no requirement is a "
|
|
8029
|
+
"conditional-acceptance condition"
|
|
8030
|
+
)
|
|
8031
|
+
+ "."
|
|
8032
|
+
)
|
|
8033
|
+
continue
|
|
8034
|
+
missing = sorted(cited - known)
|
|
8035
|
+
if missing:
|
|
8036
|
+
failures.append(
|
|
8037
|
+
f"final-verification: addedSurfaceAudit {row_id} cites "
|
|
8038
|
+
f"{missing}, which the report does not carry."
|
|
8039
|
+
)
|
|
8040
|
+
|
|
8041
|
+
|
|
7917
8042
|
def _validate_final_verification_consistency(data: dict, failures: list[str]) -> None:
|
|
7918
8043
|
"""Enforce verdict ↔ blocker/condition/routing consistency on the
|
|
7919
8044
|
final-verification data.json (SSOT). The schema guarantees field SHAPE;
|
|
@@ -7923,6 +8048,7 @@ def _validate_final_verification_consistency(data: dict, failures: list[str]) ->
|
|
|
7923
8048
|
"""
|
|
7924
8049
|
if (data.get("header") or {}).get("taskType") != "final-verification":
|
|
7925
8050
|
return
|
|
8051
|
+
_validate_added_surface_audit(data, failures)
|
|
7926
8052
|
verdict = data.get("finalVerdict") or {}
|
|
7927
8053
|
token = (verdict.get("verdictToken") or "").strip().lower()
|
|
7928
8054
|
fv = data.get("finalVerification") or {}
|
|
@@ -7954,14 +8080,19 @@ def _validate_final_verification_consistency(data: dict, failures: list[str]) ->
|
|
|
7954
8080
|
"final-verification: verdict `conditional-accept` but "
|
|
7955
8081
|
"conditionalAcceptanceConditions is empty — list every condition."
|
|
7956
8082
|
)
|
|
7957
|
-
if (
|
|
7958
|
-
|
|
7959
|
-
|
|
7960
|
-
|
|
8083
|
+
if routing_token in RELEASE_HANDOFF_TARGETS and not release_handoff_allowed(data):
|
|
8084
|
+
blocking = blocking_condition_ids(data)
|
|
8085
|
+
reason = (
|
|
8086
|
+
f"condition(s) {blocking} declare `blocksReleaseHandoff: true` "
|
|
8087
|
+
"(a condition with no declaration counts as blocking)"
|
|
8088
|
+
if blocking
|
|
8089
|
+
else f"verdict is `{token}`"
|
|
8090
|
+
)
|
|
7961
8091
|
failures.append(
|
|
7962
8092
|
f"final-verification: routingRecommendation cites `release-handoff` "
|
|
7963
|
-
f"but
|
|
7964
|
-
"
|
|
8093
|
+
f"but {reason} — release-handoff routing needs an `accepted` verdict, "
|
|
8094
|
+
"or a `conditional-accept` whose every condition declares "
|
|
8095
|
+
"`blocksReleaseHandoff: false`."
|
|
7965
8096
|
)
|
|
7966
8097
|
|
|
7967
8098
|
scope = data.get("verificationScope", "whole-task")
|
|
@@ -422,11 +422,8 @@ def _validate_analysis_verdict(
|
|
|
422
422
|
expected = "blocked" if worker_blocked or scope_blocked else (
|
|
423
423
|
"analysis-partial" if partial else "analysis-complete"
|
|
424
424
|
)
|
|
425
|
-
actual = {
|
|
426
|
-
|
|
427
|
-
str((data.get("finalVerdict") or {}).get("verdictToken") or ""),
|
|
428
|
-
}
|
|
429
|
-
if actual == {expected}:
|
|
425
|
+
actual = str((data.get("finalVerdict") or {}).get("verdictToken") or "")
|
|
426
|
+
if actual == expected:
|
|
430
427
|
return
|
|
431
428
|
reasons = []
|
|
432
429
|
if worker_blocked:
|
|
@@ -1034,7 +1034,7 @@ def _check_batch_cleanup_checkpoints(
|
|
|
1034
1034
|
실제로 일어났는지 (prompts/profiles/_common-contract.md 'Phase-start cleanup').
|
|
1035
1035
|
R1=convergence round 1 직전, R2=report-writer dispatch 직전(수렴이 있었으면
|
|
1036
1036
|
마지막 라운드 이후 별도 1회). ISO-8601 ts 는 lexicographic 비교가 곧 시간순."""
|
|
1037
|
-
detail = "prompts/
|
|
1037
|
+
detail = "prompts/lead/okstra-lead-contract.md 'Run-scoped worker-resource lifecycle'"
|
|
1038
1038
|
cleanup_ts = sorted(ts for ts, _line in by_phase.get("phase-batch-cleanup", []))
|
|
1039
1039
|
conv_ts = sorted(ts for ts, _line in by_phase.get("phase-5.5-convergence", []))
|
|
1040
1040
|
synth_ts = sorted(ts for ts, _line in by_phase.get("phase-6-synthesis", []))
|