okstra 0.159.0 → 0.161.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/docs/architecture/storage-model.md +2 -0
- package/docs/architecture.md +2 -1
- package/docs/cli.md +8 -3
- package/docs/for-ai/README.md +2 -2
- package/docs/for-ai/skills/okstra-inspect.md +3 -0
- package/docs/for-ai/skills/okstra-run.md +2 -1
- package/docs/for-ai/skills/okstra-user-response.md +5 -5
- package/docs/project-structure-overview.md +5 -1
- package/docs/task-process/implementation.md +28 -0
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/bin/okstra-claude-exec.sh +4 -1
- package/runtime/prompts/host-orchestration/README.md +18 -0
- package/runtime/prompts/host-orchestration/implementation.md +57 -0
- package/runtime/prompts/launch.template.md +10 -1
- package/runtime/prompts/lead/adapters/claude-code.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +67 -0
- package/runtime/prompts/lead/context-loader.md +5 -2
- package/runtime/prompts/lead/convergence.md +3 -1
- package/runtime/prompts/lead/plan-body-verification.md +21 -2
- package/runtime/prompts/lead/team-contract.md +2 -1
- package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
- package/runtime/prompts/profiles/_common-contract.md +3 -1
- package/runtime/prompts/profiles/implementation-planning.md +2 -0
- package/runtime/prompts/profiles/requirements-discovery.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +3 -0
- package/runtime/python/okstra_ctl/clarification_items.py +9 -0
- package/runtime/python/okstra_ctl/cmux.py +531 -0
- package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
- package/runtime/python/okstra_ctl/convergence.py +168 -11
- package/runtime/python/okstra_ctl/dispatch_core.py +76 -7
- package/runtime/python/okstra_ctl/dispatch_state.py +16 -0
- package/runtime/python/okstra_ctl/error_issue.py +640 -0
- package/runtime/python/okstra_ctl/error_report.py +56 -0
- package/runtime/python/okstra_ctl/error_zip.py +23 -10
- package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
- package/runtime/python/okstra_ctl/issue_signals.py +186 -0
- package/runtime/python/okstra_ctl/lead_runtime.py +30 -2
- package/runtime/python/okstra_ctl/paths.py +38 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
- package/runtime/python/okstra_ctl/profile_show.py +134 -0
- package/runtime/python/okstra_ctl/recap.py +63 -0
- package/runtime/python/okstra_ctl/render.py +7 -2
- package/runtime/python/okstra_ctl/render_final_report.py +7 -22
- package/runtime/python/okstra_ctl/report_translation.py +4 -0
- package/runtime/python/okstra_ctl/report_views.py +7 -3
- package/runtime/python/okstra_ctl/run.py +54 -3
- package/runtime/python/okstra_ctl/run_audit.py +477 -0
- package/runtime/python/okstra_ctl/team.py +50 -11
- package/runtime/python/okstra_ctl/user_response.py +25 -10
- package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
- package/runtime/python/okstra_ctl/wizard.py +64 -10
- package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
- package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
- package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
- package/runtime/schemas/final-report-v1.0.schema.json +14 -0
- package/runtime/schemas/final-report-v2.0.schema.json +51 -1
- package/runtime/skills/okstra-inspect/SKILL.md +3 -1
- package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
- package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
- package/runtime/skills/okstra-run/SKILL.md +28 -10
- package/runtime/skills/okstra-user-response/SKILL.md +18 -18
- package/runtime/templates/reports/final-report.template.md +4 -0
- package/runtime/templates/reports/html/i18n/en.json +5 -1
- package/runtime/templates/reports/html/i18n/ko.json +5 -1
- package/runtime/templates/reports/html/macros/forms.html +15 -0
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
- package/runtime/templates/reports/i18n/en.json +2 -0
- package/runtime/validators/validate-run.py +267 -208
- package/runtime/validators/validate-workflow.sh +6 -0
- package/runtime/validators/validate_session_conformance.py +135 -31
- package/src/cli-registry.mjs +34 -0
- package/src/commands/execute/incremental-scope.mjs +10 -0
- package/src/commands/execute/worker-audit-check.mjs +35 -0
- package/src/commands/inspect/error-issue.mjs +27 -0
- package/src/commands/inspect/profile-show.mjs +29 -0
- package/src/commands/inspect/run-audit.mjs +26 -0
|
@@ -66,6 +66,7 @@ from okstra_ctl.report_translation import ( # noqa: E402
|
|
|
66
66
|
hangul_share,
|
|
67
67
|
)
|
|
68
68
|
from okstra_ctl.stage_citations import cited_stage_numbers # noqa: E402
|
|
69
|
+
from okstra_ctl.incremental_scope import coverage_row_blocked_on # noqa: E402
|
|
69
70
|
from okstra_ctl.workflow import DEFAULT_NEXT_PHASE, PHASE_SEQUENCE # noqa: E402
|
|
70
71
|
from okstra_ctl.md_table import ( # noqa: E402
|
|
71
72
|
is_separator_row as _is_markdown_separator,
|
|
@@ -103,7 +104,10 @@ from okstra_ctl.worker_prompt_contract import ( # noqa: E402
|
|
|
103
104
|
PromptRecord,
|
|
104
105
|
validate_initial_prompt_records,
|
|
105
106
|
)
|
|
106
|
-
from okstra_ctl.
|
|
107
|
+
from okstra_ctl.worker_audit_ledger import ( # noqa: E402
|
|
108
|
+
READING_CONFIRMATION_HEADING_RE,
|
|
109
|
+
check_worker_results_audit,
|
|
110
|
+
)
|
|
107
111
|
from validate_analysis_report import validate_analysis_report # noqa: E402
|
|
108
112
|
from okstra_ctl.convergence_engine import ( # noqa: E402
|
|
109
113
|
grouped_input_digest,
|
|
@@ -1082,7 +1086,7 @@ def _validate_v2_ai_handoff(content: str, failures: list[str]) -> None:
|
|
|
1082
1086
|
"schema-v2 AI handoff markdown contains human-only field "
|
|
1083
1087
|
f"{field!r}; render it only in the task-specific HTML."
|
|
1084
1088
|
)
|
|
1085
|
-
if
|
|
1089
|
+
if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
|
|
1086
1090
|
failures.append(
|
|
1087
1091
|
"final report contains a `## 0. Reading Confirmation` heading — "
|
|
1088
1092
|
"Reading Confirmation lives in the worker audit sidecar, never "
|
|
@@ -1102,12 +1106,6 @@ _REPORT_INDEX_ANCHOR_RE = re.compile(r'<a id="report-index"')
|
|
|
1102
1106
|
# never anchored — i.e. an un-anchored ID that the Index cannot link to.
|
|
1103
1107
|
_UNANCHORED_ID_ROW_RE = re.compile(r"^\|[ \t]*\*{0,2}([A-Z]{1,4}-\d{3,})\b", re.MULTILINE)
|
|
1104
1108
|
|
|
1105
|
-
# Reading Confirmation heading must NOT appear in the final-report — it
|
|
1106
|
-
# belongs in the worker audit sidecar (`<worker>-audit-<task-type>-<seq>.md`).
|
|
1107
|
-
_READING_CONFIRMATION_HEADING_RE = re.compile(
|
|
1108
|
-
r"^##[ \t]+0\.[ \t]+Reading Confirmation\b", re.MULTILINE
|
|
1109
|
-
)
|
|
1110
|
-
|
|
1111
1109
|
# Empty Section 0 (Clarification Response Carried In) stub. When no
|
|
1112
1110
|
# carry-in path is provided, the writer must OMIT the `## 0.` heading
|
|
1113
1111
|
# entirely — emitting the heading followed by the "No prior clarification
|
|
@@ -2349,7 +2347,7 @@ def validate_report(
|
|
|
2349
2347
|
|
|
2350
2348
|
# Reading Confirmation belongs in the worker audit sidecar, not the
|
|
2351
2349
|
# user-facing final-report.
|
|
2352
|
-
if
|
|
2350
|
+
if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
|
|
2353
2351
|
failures.append(
|
|
2354
2352
|
"final report contains a `## 0. Reading Confirmation` heading — "
|
|
2355
2353
|
"Reading Confirmation lives in the worker audit sidecar "
|
|
@@ -2385,27 +2383,6 @@ def validate_report(
|
|
|
2385
2383
|
failures.append(f"final report contains {remedy}")
|
|
2386
2384
|
|
|
2387
2385
|
|
|
2388
|
-
# Worker-results filename pattern: `<worker-role>-<task-type>-<seq>.md`.
|
|
2389
|
-
# Every analysis-worker role name ends in `-worker` (`claude-worker`,
|
|
2390
|
-
# `codex-worker`, `antigravity-worker`, `report-writer-worker`), so anchor the
|
|
2391
|
-
# split on that suffix — otherwise `antigravity-worker-error-analysis-001.md`
|
|
2392
|
-
# ambiguously parses as `worker=antigravity, task=worker-error-analysis`.
|
|
2393
|
-
# Audit sidecars (`*-audit-*`) and errors sidecars (`.json`) are not
|
|
2394
|
-
# matched here.
|
|
2395
|
-
_WORKER_RESULT_BASENAME_RE = re.compile(
|
|
2396
|
-
r"^(?P<worker>[a-z][a-z0-9-]*-worker)-(?P<task_type>[a-z][a-z-]*?)-(?P<seq>\d{3})\.md$"
|
|
2397
|
-
)
|
|
2398
|
-
_EVIDENCE_READ_RE = re.compile(
|
|
2399
|
-
r"^- Evidence read: `(?P<path>[^`\n]+)`\s*$",
|
|
2400
|
-
re.MULTILINE,
|
|
2401
|
-
)
|
|
2402
|
-
_FILE_LINE_CITATION_RE = re.compile(
|
|
2403
|
-
r"`(?P<path>(?!https?://)[^`\n]+?):(?P<line>\d+(?:-\d+)?)`"
|
|
2404
|
-
)
|
|
2405
|
-
_EXTENSIONLESS_SOURCE_FILENAMES = frozenset(
|
|
2406
|
-
{"Dockerfile", "Justfile", "Makefile", "Procfile", "Rakefile"}
|
|
2407
|
-
)
|
|
2408
|
-
|
|
2409
2386
|
_REPORT_BASENAME_SEQ_RE = re.compile(r"-(?P<seq>\d{3})(?:\.data)?\.(?:md|json)$")
|
|
2410
2387
|
|
|
2411
2388
|
|
|
@@ -2417,181 +2394,23 @@ def _report_run_seq(report_path: Path) -> str | None:
|
|
|
2417
2394
|
return match.group("seq") if match else None
|
|
2418
2395
|
|
|
2419
2396
|
|
|
2420
|
-
def _cited_file_paths(content: str) -> set[str]:
|
|
2421
|
-
paths: set[str] = set()
|
|
2422
|
-
for match in _FILE_LINE_CITATION_RE.finditer(content):
|
|
2423
|
-
path = match.group("path")
|
|
2424
|
-
if _looks_like_file_path(path):
|
|
2425
|
-
paths.add(path)
|
|
2426
|
-
return paths
|
|
2427
|
-
|
|
2428
|
-
|
|
2429
|
-
def _looks_like_file_path(path: str) -> bool:
|
|
2430
|
-
if (
|
|
2431
|
-
not path
|
|
2432
|
-
or path.startswith(("-", "$"))
|
|
2433
|
-
or any(char.isspace() for char in path)
|
|
2434
|
-
):
|
|
2435
|
-
return False
|
|
2436
|
-
if re.fullmatch(r"[0-9a-fA-F]{7,64}", path):
|
|
2437
|
-
return False
|
|
2438
|
-
return (
|
|
2439
|
-
"/" in path
|
|
2440
|
-
or "." in Path(path).name
|
|
2441
|
-
or Path(path).name in _EXTENSIONLESS_SOURCE_FILENAMES
|
|
2442
|
-
)
|
|
2443
|
-
|
|
2444
|
-
|
|
2445
|
-
def _audit_evidence_read_paths(content: str) -> set[str]:
|
|
2446
|
-
return {
|
|
2447
|
-
match.group("path")
|
|
2448
|
-
for match in _EVIDENCE_READ_RE.finditer(content)
|
|
2449
|
-
}
|
|
2450
|
-
|
|
2451
|
-
|
|
2452
|
-
def _worker_prompt_path(
|
|
2453
|
-
report_path: Path,
|
|
2454
|
-
worker_role: str,
|
|
2455
|
-
task_type: str,
|
|
2456
|
-
seq: str,
|
|
2457
|
-
) -> Path:
|
|
2458
|
-
return (
|
|
2459
|
-
report_path.parent.parent
|
|
2460
|
-
/ "prompts"
|
|
2461
|
-
/ f"{worker_role}-prompt-{task_type}-{seq}.md"
|
|
2462
|
-
)
|
|
2463
|
-
|
|
2464
|
-
|
|
2465
|
-
def _validate_worker_evidence_read_ledger(
|
|
2466
|
-
*,
|
|
2467
|
-
report_path: Path,
|
|
2468
|
-
worker_role: str,
|
|
2469
|
-
task_type: str,
|
|
2470
|
-
seq: str,
|
|
2471
|
-
result_name: str,
|
|
2472
|
-
result_content: str,
|
|
2473
|
-
audit_path: Path,
|
|
2474
|
-
failures: list[str],
|
|
2475
|
-
) -> None:
|
|
2476
|
-
if worker_role == "report-writer-worker":
|
|
2477
|
-
return
|
|
2478
|
-
prompt_path = _worker_prompt_path(report_path, worker_role, task_type, seq)
|
|
2479
|
-
try:
|
|
2480
|
-
prompt_content = prompt_path.read_text(encoding="utf-8")
|
|
2481
|
-
except OSError:
|
|
2482
|
-
return
|
|
2483
|
-
if EVIDENCE_LEDGER_HEADER not in prompt_content.splitlines():
|
|
2484
|
-
return
|
|
2485
|
-
try:
|
|
2486
|
-
audit_content = audit_path.read_text(encoding="utf-8")
|
|
2487
|
-
except OSError as exc:
|
|
2488
|
-
failures.append(
|
|
2489
|
-
f"worker audit sidecar unreadable: {audit_path.name} ({exc})"
|
|
2490
|
-
)
|
|
2491
|
-
return
|
|
2492
|
-
|
|
2493
|
-
missing_paths = sorted(
|
|
2494
|
-
_cited_file_paths(result_content) - _audit_evidence_read_paths(audit_content)
|
|
2495
|
-
)
|
|
2496
|
-
for missing_path in missing_paths:
|
|
2497
|
-
failures.append(
|
|
2498
|
-
f"worker `{worker_role}` result `{result_name}` cites "
|
|
2499
|
-
f"`{missing_path}:line` without an Evidence read row for "
|
|
2500
|
-
f"`{missing_path}` in `{audit_path.name}`"
|
|
2501
|
-
)
|
|
2502
|
-
|
|
2503
|
-
|
|
2504
2397
|
def validate_worker_results_audit(
|
|
2505
2398
|
report_path: Path, task_type: str, failures: list[str]
|
|
2506
2399
|
) -> None:
|
|
2507
|
-
"""Enforce the worker audit sidecar contract.
|
|
2508
|
-
|
|
2509
|
-
For every `worker-results/<worker>-<task-type>-<seq>.md` **this run**
|
|
2510
|
-
produced (skipping the audit sidecar itself), the validator checks:
|
|
2511
|
-
|
|
2512
|
-
1. The main worker-results file does NOT contain a `## 0. Reading
|
|
2513
|
-
Confirmation` heading. That block moved to the audit sidecar with
|
|
2514
|
-
the report-format readability pass.
|
|
2515
|
-
2. The matching audit sidecar exists at
|
|
2516
|
-
`<worker>-audit-<task-type>-<seq>.md`. Missing sidecar means the
|
|
2517
|
-
worker silently skipped the reading-confirmation step.
|
|
2518
|
-
3. For new prompts carrying the required-v1 marker, every canonical
|
|
2519
|
-
backticked `path:line` citation has a matching Evidence read row in
|
|
2520
|
-
that audit sidecar. Historical prompts without the marker retain the
|
|
2521
|
-
existence-only contract.
|
|
2522
|
-
|
|
2523
|
-
Scoped to this run's seq. `worker-results/` accumulates every run's
|
|
2524
|
-
artifacts, so scanning the whole directory judged a run by files it did
|
|
2525
|
-
not produce: a task that once opted a worker in and later dropped it from
|
|
2526
|
-
the roster failed forever on that worker's old sidecar, with no legitimate
|
|
2527
|
-
remedy — the lead can neither fabricate a sidecar for a worker it never
|
|
2528
|
-
dispatched nor delete a prior run's audit record.
|
|
2529
|
-
"""
|
|
2530
|
-
# `report_path` is `runs/<task-type>/reports/final-report-...md`; the
|
|
2531
|
-
# sibling `worker-results/` directory holds every worker artifact.
|
|
2532
|
-
worker_results_dir = report_path.parent.parent / "worker-results"
|
|
2533
|
-
if not worker_results_dir.is_dir():
|
|
2534
|
-
# No worker-results directory means no analysis workers ran (e.g.
|
|
2535
|
-
# `release-handoff` which is single-lead). Nothing to enforce.
|
|
2536
|
-
return
|
|
2400
|
+
"""Enforce the worker audit sidecar contract at Phase 7.
|
|
2537
2401
|
|
|
2538
|
-
|
|
2539
|
-
|
|
2540
|
-
|
|
2541
|
-
|
|
2542
|
-
|
|
2543
|
-
|
|
2544
|
-
|
|
2545
|
-
|
|
2546
|
-
|
|
2547
|
-
|
|
2548
|
-
|
|
2549
|
-
|
|
2550
|
-
# Cross-phase artifacts shouldn't appear here; skip rather
|
|
2551
|
-
# than fail to keep the check focused on the current phase.
|
|
2552
|
-
continue
|
|
2553
|
-
if run_seq is not None and match.group("seq") != run_seq:
|
|
2554
|
-
# A prior run's artifact. Its contract was judged when it ran.
|
|
2555
|
-
continue
|
|
2556
|
-
|
|
2557
|
-
worker_role = match.group("worker")
|
|
2558
|
-
seq = match.group("seq")
|
|
2559
|
-
rel = path.name
|
|
2560
|
-
try:
|
|
2561
|
-
content = path.read_text()
|
|
2562
|
-
except OSError as exc:
|
|
2563
|
-
failures.append(f"worker-results file unreadable: {rel} ({exc})")
|
|
2564
|
-
continue
|
|
2565
|
-
|
|
2566
|
-
if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
|
|
2567
|
-
failures.append(
|
|
2568
|
-
f"worker-results file `{rel}` contains a `## 0. Reading "
|
|
2569
|
-
f"Confirmation` heading — that block moved to the audit "
|
|
2570
|
-
f"sidecar (`{worker_role}-audit-{task_type}-{seq}.md`). "
|
|
2571
|
-
f"Remove the §0 heading + body from the main file and "
|
|
2572
|
-
f"write a fresh sidecar."
|
|
2573
|
-
)
|
|
2574
|
-
|
|
2575
|
-
audit_path = worker_results_dir / f"{worker_role}-audit-{task_type}-{seq}.md"
|
|
2576
|
-
if not audit_path.exists():
|
|
2577
|
-
failures.append(
|
|
2578
|
-
f"worker `{worker_role}` produced `{rel}` but no audit sidecar "
|
|
2579
|
-
f"at `{audit_path.name}` — the sidecar must carry the Reading "
|
|
2580
|
-
f"Confirmation block (one short line per input file). Workers "
|
|
2581
|
-
f"write this in the same step as the main worker-results file."
|
|
2582
|
-
)
|
|
2583
|
-
continue
|
|
2584
|
-
|
|
2585
|
-
_validate_worker_evidence_read_ledger(
|
|
2586
|
-
report_path=report_path,
|
|
2587
|
-
worker_role=worker_role,
|
|
2588
|
-
task_type=task_type,
|
|
2589
|
-
seq=seq,
|
|
2590
|
-
result_name=rel,
|
|
2591
|
-
result_content=content,
|
|
2592
|
-
audit_path=audit_path,
|
|
2593
|
-
failures=failures,
|
|
2594
|
-
)
|
|
2402
|
+
The rules themselves live in `okstra_ctl.worker_audit_ledger` so that
|
|
2403
|
+
`okstra worker-audit-check` can apply the identical checks mid-run, while
|
|
2404
|
+
the worker session is still alive and can fix its own citations. All this
|
|
2405
|
+
wrapper adds is the Phase 7 anchor: the run directory and this run's seq,
|
|
2406
|
+
both read off the report path.
|
|
2407
|
+
"""
|
|
2408
|
+
failures.extend(check_worker_results_audit(
|
|
2409
|
+
# `report_path` is `runs/<task-type>/reports/final-report-...md`.
|
|
2410
|
+
report_path.parent.parent,
|
|
2411
|
+
task_type,
|
|
2412
|
+
_report_run_seq(report_path),
|
|
2413
|
+
))
|
|
2595
2414
|
|
|
2596
2415
|
|
|
2597
2416
|
def validate_team_state_usage(team_state: dict, failures: list[str]) -> None:
|
|
@@ -3307,6 +3126,9 @@ def validate_final_report_data(
|
|
|
3307
3126
|
_validate_verdict_card_fields(data, failures)
|
|
3308
3127
|
# Phase-agnostic: the coverage critic runs in every finding-producing phase.
|
|
3309
3128
|
_validate_unverified_critic_gaps_recorded(data, failures)
|
|
3129
|
+
# Called here rather than from a task-type branch: four profiles raise
|
|
3130
|
+
# clarification rows, and the gate scopes itself by task type internally.
|
|
3131
|
+
_validate_clarification_options(data, failures)
|
|
3310
3132
|
|
|
3311
3133
|
task_type = (data.get("header") or {}).get("taskType")
|
|
3312
3134
|
_validate_verifier_fail_blocks_verdict(data, failures)
|
|
@@ -3331,6 +3153,8 @@ def validate_final_report_data(
|
|
|
3331
3153
|
print(f"validate-run: warning: {warning}", file=sys.stderr)
|
|
3332
3154
|
_validate_supersession_ledger(data, failures)
|
|
3333
3155
|
_validate_clarification_evidence_note(data, failures)
|
|
3156
|
+
_validate_approval_clarification_backtrace(data, failures)
|
|
3157
|
+
_validate_rerun_guidance(data, failures)
|
|
3334
3158
|
_validate_variation_point_analysis(
|
|
3335
3159
|
(data.get("implementationPlanning") or {}).get("variationPointAnalysis"),
|
|
3336
3160
|
resolve_architecture(_project_root_from_report(report_path)),
|
|
@@ -3658,6 +3482,38 @@ def _self_fix_budget_exhausted(pbv: dict) -> bool:
|
|
|
3658
3482
|
)
|
|
3659
3483
|
|
|
3660
3484
|
|
|
3485
|
+
def _state_classification(item: dict, gate_class: str) -> str:
|
|
3486
|
+
"""This item's `planItems[].rounds[].classification` for the state file.
|
|
3487
|
+
|
|
3488
|
+
The gate classifier deliberately folds `partial-consensus` and
|
|
3489
|
+
`dissent-isolated` into `has-dissent` — only the majority-disagree boundary
|
|
3490
|
+
moves the gate. The state file records the finer label, and the information
|
|
3491
|
+
to recover it is in the same verdicts, so the mapping lives beside the
|
|
3492
|
+
classifier rather than being re-invented by each lead.
|
|
3493
|
+
|
|
3494
|
+
*gate_class* is passed in rather than recomputed so that the caller's
|
|
3495
|
+
effective classification — which may have been downgraded by
|
|
3496
|
+
`_is_dissent_downgraded` — is the one this translates.
|
|
3497
|
+
|
|
3498
|
+
`contested` never appears: it is only meaningful at `maxRounds > 1`, and at
|
|
3499
|
+
the default `maxRounds=1` the round protocol folds any otherwise-unresolved
|
|
3500
|
+
item into `partial-consensus`.
|
|
3501
|
+
"""
|
|
3502
|
+
if gate_class == "all-non-result":
|
|
3503
|
+
# No non-error vote at all is the `needs-reverify` shape taken to its
|
|
3504
|
+
# limit — "fewer than 2 participating votes" covers zero.
|
|
3505
|
+
return "needs-reverify"
|
|
3506
|
+
if gate_class != "has-dissent":
|
|
3507
|
+
return gate_class
|
|
3508
|
+
dissenting = sum(
|
|
3509
|
+
1
|
|
3510
|
+
for vote in (item.get("verdicts") or [])
|
|
3511
|
+
if isinstance(vote, dict)
|
|
3512
|
+
and str(vote.get("verdict") or "").strip().upper() == "DISAGREE"
|
|
3513
|
+
)
|
|
3514
|
+
return "dissent-isolated" if dissenting == 1 else "partial-consensus"
|
|
3515
|
+
|
|
3516
|
+
|
|
3661
3517
|
def _is_dissent_downgraded(item: dict, pbv: dict) -> bool:
|
|
3662
3518
|
"""Whether a surviving `majority-disagree` item stops blocking approval.
|
|
3663
3519
|
|
|
@@ -4180,6 +4036,200 @@ def _validate_clarification_evidence_note(data: dict, failures: list[str]) -> No
|
|
|
4180
4036
|
)
|
|
4181
4037
|
|
|
4182
4038
|
|
|
4039
|
+
_CLARIFICATION_OPTION_SCHEMA_VERSION = "2.0"
|
|
4040
|
+
# The four profiles that read `_clarification-recommendation.md`. Unlike the
|
|
4041
|
+
# evidence-note gate above — which is called from inside the
|
|
4042
|
+
# `implementation-planning` branch — this one is called phase-agnostically, so
|
|
4043
|
+
# the task-type filter here is the only thing scoping it.
|
|
4044
|
+
_CLARIFICATION_OPTION_TASK_TYPES = frozenset({
|
|
4045
|
+
"error-analysis",
|
|
4046
|
+
"implementation-planning",
|
|
4047
|
+
"improvement-discovery",
|
|
4048
|
+
"requirements-discovery",
|
|
4049
|
+
})
|
|
4050
|
+
# `in-repo` and `cross-repo` answer the same question, so exactly one may hold.
|
|
4051
|
+
_REACH_TOKENS = frozenset({"in-repo", "cross-repo"})
|
|
4052
|
+
_LEGACY_EXPECTED_FORM_RE = re.compile(r"\b(?:Recommended|Alternatives):")
|
|
4053
|
+
|
|
4054
|
+
|
|
4055
|
+
def _validate_clarification_options(data: dict, failures: list[str]) -> None:
|
|
4056
|
+
"""A `decision` row must carry its choices as data, not as prose.
|
|
4057
|
+
|
|
4058
|
+
The choices used to live inside the `expectedForm` string, where two
|
|
4059
|
+
separate parsers split them differently and neither was checked — the board
|
|
4060
|
+
the user picked from could disagree with the board the report meant.
|
|
4061
|
+
Structured options remove the parsing; this gate keeps the structure
|
|
4062
|
+
honest. Whether an impact claim is *true* is the adversarial round's job,
|
|
4063
|
+
exactly as with `Evidence checked:`.
|
|
4064
|
+
|
|
4065
|
+
schema-v1 is exempt because it cannot comply: it keeps clarifications as a
|
|
4066
|
+
Markdown table of strings and its schema forbids an `options` property, so
|
|
4067
|
+
demanding one would fail every v1 `decision` row for a structure the format
|
|
4068
|
+
has no place to hold.
|
|
4069
|
+
"""
|
|
4070
|
+
if data.get("schemaVersion") != _CLARIFICATION_OPTION_SCHEMA_VERSION:
|
|
4071
|
+
return
|
|
4072
|
+
task_type = (data.get("header") or {}).get("taskType")
|
|
4073
|
+
if task_type not in _CLARIFICATION_OPTION_TASK_TYPES:
|
|
4074
|
+
return
|
|
4075
|
+
for row in data.get("clarificationItems") or []:
|
|
4076
|
+
if not isinstance(row, dict) or row.get("kind") != "decision":
|
|
4077
|
+
continue
|
|
4078
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4079
|
+
_validate_option_set(row.get("options"), row_id, failures)
|
|
4080
|
+
if _LEGACY_EXPECTED_FORM_RE.search(str(row.get("expectedForm") or "")):
|
|
4081
|
+
failures.append(
|
|
4082
|
+
f"final-report data.json: clarification `{row_id}` still encodes "
|
|
4083
|
+
"its choices in `expectedForm` (`Recommended:` / `Alternatives:`). "
|
|
4084
|
+
"Choices belong in `options[]`; `expectedForm` states only the "
|
|
4085
|
+
"shape of the answer. Two sources for one fact leave consumers "
|
|
4086
|
+
"disagreeing about which is authoritative."
|
|
4087
|
+
)
|
|
4088
|
+
|
|
4089
|
+
|
|
4090
|
+
def _validate_option_set(
|
|
4091
|
+
options: object, row_id: str, failures: list[str]
|
|
4092
|
+
) -> None:
|
|
4093
|
+
"""Check one `decision` row's `options[]` for pickability and reach."""
|
|
4094
|
+
if not isinstance(options, list) or len(options) < 2:
|
|
4095
|
+
failures.append(
|
|
4096
|
+
f"final-report data.json: clarification `{row_id}` is a `decision` "
|
|
4097
|
+
"but does not offer at least two `options[]`. A decision the user "
|
|
4098
|
+
"cannot choose between is not a decision."
|
|
4099
|
+
)
|
|
4100
|
+
return
|
|
4101
|
+
entries = [option for option in options if isinstance(option, dict)]
|
|
4102
|
+
recommended = sum(1 for option in entries if option.get("role") == "recommended")
|
|
4103
|
+
if recommended != 1:
|
|
4104
|
+
failures.append(
|
|
4105
|
+
f"final-report data.json: clarification `{row_id}` must carry exactly "
|
|
4106
|
+
f"one `role: recommended` option (found {recommended}). The reader "
|
|
4107
|
+
"needs to know which answer the run stands behind."
|
|
4108
|
+
)
|
|
4109
|
+
for index, option in enumerate(entries):
|
|
4110
|
+
tokens = option.get("scopeImpact")
|
|
4111
|
+
reach = (
|
|
4112
|
+
[token for token in tokens if token in _REACH_TOKENS]
|
|
4113
|
+
if isinstance(tokens, list)
|
|
4114
|
+
else []
|
|
4115
|
+
)
|
|
4116
|
+
if len(reach) != 1:
|
|
4117
|
+
failures.append(
|
|
4118
|
+
f"final-report data.json: clarification `{row_id}` option "
|
|
4119
|
+
f"[{index}] must declare exactly one reach token — `in-repo` or "
|
|
4120
|
+
f"`cross-repo` (found {len(reach)}). The reader cannot weigh an "
|
|
4121
|
+
"option whose reach is unstated or self-contradictory."
|
|
4122
|
+
)
|
|
4123
|
+
|
|
4124
|
+
|
|
4125
|
+
def _has_clarification_backtrace(
|
|
4126
|
+
row_id: str, plan_items: object, coverage: object
|
|
4127
|
+
) -> bool:
|
|
4128
|
+
"""Whether the plan records anything this clarification blocks.
|
|
4129
|
+
|
|
4130
|
+
Two link shapes, both authored by the same run: the `P-*` plan item that
|
|
4131
|
+
carries the `clarificationId`, and the requirement-coverage row blocked on
|
|
4132
|
+
the id. `incremental-scope` resolves impacted stages from exactly these
|
|
4133
|
+
two, and the coverage side goes through its predicate so the gate and the
|
|
4134
|
+
resolver cannot disagree about what counts as a link.
|
|
4135
|
+
"""
|
|
4136
|
+
if isinstance(plan_items, list) and any(
|
|
4137
|
+
isinstance(item, dict) and item.get("clarificationId") == row_id
|
|
4138
|
+
for item in plan_items
|
|
4139
|
+
):
|
|
4140
|
+
return True
|
|
4141
|
+
return isinstance(coverage, list) and any(
|
|
4142
|
+
coverage_row_blocked_on(row, row_id) for row in coverage
|
|
4143
|
+
)
|
|
4144
|
+
|
|
4145
|
+
|
|
4146
|
+
def _validate_approval_clarification_backtrace(
|
|
4147
|
+
data: dict, failures: list[str]
|
|
4148
|
+
) -> None:
|
|
4149
|
+
"""An approval blocker must record what it blocks.
|
|
4150
|
+
|
|
4151
|
+
`_validate_plan_body_clarification_matching` already walks the other
|
|
4152
|
+
direction — a majority-disagree plan item must cite a `blocks: approval`
|
|
4153
|
+
row. Nothing walked this way, so a row could withhold approval while
|
|
4154
|
+
recording no blast radius at all. The cost lands on the re-run:
|
|
4155
|
+
`incremental-scope` resolves impacted stages from these links and treats an
|
|
4156
|
+
id that traces to no stage as grounds to re-verify everything, so one
|
|
4157
|
+
unlinked blocker turns a narrow re-run into a full one.
|
|
4158
|
+
"""
|
|
4159
|
+
if (data.get("header") or {}).get("taskType") != "implementation-planning":
|
|
4160
|
+
return
|
|
4161
|
+
planning = data.get("implementationPlanning")
|
|
4162
|
+
if not isinstance(planning, dict):
|
|
4163
|
+
return
|
|
4164
|
+
coverage = planning.get("requirementCoverage")
|
|
4165
|
+
verification = planning.get("planBodyVerification")
|
|
4166
|
+
plan_items = (
|
|
4167
|
+
verification.get("planItems") if isinstance(verification, dict) else None
|
|
4168
|
+
)
|
|
4169
|
+
for row in data.get("clarificationItems") or []:
|
|
4170
|
+
if not isinstance(row, dict) or row.get("blocks") != "approval":
|
|
4171
|
+
continue
|
|
4172
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4173
|
+
if _has_clarification_backtrace(row_id, plan_items, coverage):
|
|
4174
|
+
continue
|
|
4175
|
+
failures.append(
|
|
4176
|
+
f"final-report data.json: clarification `{row_id}` blocks approval "
|
|
4177
|
+
"but has no back-trace into the plan — no plan item carries it as "
|
|
4178
|
+
"`clarificationId`, and no requirement-coverage row is `blocked "
|
|
4179
|
+
f"{row_id}` in its `status` or `approvalDisposition`. An item that "
|
|
4180
|
+
"withholds approval without recording what it affects forces the "
|
|
4181
|
+
"next re-run to re-verify everything."
|
|
4182
|
+
)
|
|
4183
|
+
|
|
4184
|
+
|
|
4185
|
+
_RERUN_FLAG = "--answered-clarifications"
|
|
4186
|
+
|
|
4187
|
+
|
|
4188
|
+
def _next_step_texts(steps: object) -> list[str]:
|
|
4189
|
+
"""Every reader-visible string in `recommendedNextSteps`, prose and command."""
|
|
4190
|
+
texts: list[str] = []
|
|
4191
|
+
for step in steps if isinstance(steps, list) else []:
|
|
4192
|
+
if not isinstance(step, dict):
|
|
4193
|
+
continue
|
|
4194
|
+
texts.append(str(step.get("text") or ""))
|
|
4195
|
+
for command in step.get("commands") or []:
|
|
4196
|
+
if isinstance(command, dict):
|
|
4197
|
+
texts.append(str(command.get("claudeCode") or ""))
|
|
4198
|
+
texts.append(str(command.get("terminal") or ""))
|
|
4199
|
+
return texts
|
|
4200
|
+
|
|
4201
|
+
|
|
4202
|
+
def _validate_rerun_guidance(data: dict, failures: list[str]) -> None:
|
|
4203
|
+
"""A report that withholds approval must say how to come back from it.
|
|
4204
|
+
|
|
4205
|
+
The way forward — answer the blockers, then re-run carrying those ids —
|
|
4206
|
+
lived only in the lead prompt, which is read *after* the next run has
|
|
4207
|
+
already started. The person who has to act reads the report instead, and it
|
|
4208
|
+
told them nothing about the next command. Requiring the flag by name is a
|
|
4209
|
+
low bar deliberately: it does not check that the rest of the step is right,
|
|
4210
|
+
only that the report stops leaving the reader to work the mechanics out.
|
|
4211
|
+
"""
|
|
4212
|
+
if (data.get("header") or {}).get("taskType") != "implementation-planning":
|
|
4213
|
+
return
|
|
4214
|
+
has_blocker = any(
|
|
4215
|
+
isinstance(row, dict) and row.get("blocks") == "approval"
|
|
4216
|
+
for row in data.get("clarificationItems") or []
|
|
4217
|
+
)
|
|
4218
|
+
if not has_blocker:
|
|
4219
|
+
return
|
|
4220
|
+
if any(_RERUN_FLAG in text for text in _next_step_texts(
|
|
4221
|
+
data.get("recommendedNextSteps")
|
|
4222
|
+
)):
|
|
4223
|
+
return
|
|
4224
|
+
failures.append(
|
|
4225
|
+
"final-report data.json: this plan withholds approval on a "
|
|
4226
|
+
"`blocks: approval` clarification, but no `recommendedNextSteps` entry "
|
|
4227
|
+
f"tells the reader how to resume — name the `{_RERUN_FLAG}` re-run in a "
|
|
4228
|
+
"step's `text` or one of its `commands`. `okstra recap assemble` "
|
|
4229
|
+
"prints the exact ids and flag value once the answers are recorded."
|
|
4230
|
+
)
|
|
4231
|
+
|
|
4232
|
+
|
|
4183
4233
|
def _validate_self_fix_grouping(data: dict, failures: list[str]) -> None:
|
|
4184
4234
|
"""A self-fix round must be instructed by cause, not as a flat item list.
|
|
4185
4235
|
|
|
@@ -5930,6 +5980,21 @@ def validate_plan_body_section(
|
|
|
5930
5980
|
return [*_detect_self_fix_recurrence(pbv), *_detect_uniform_verifier(pbv)]
|
|
5931
5981
|
|
|
5932
5982
|
|
|
5983
|
+
def _gate_summary_item(item: dict, pbv: dict) -> dict:
|
|
5984
|
+
"""One `gate.items[]` row: the gate class plus its state-file counterpart."""
|
|
5985
|
+
classification = (
|
|
5986
|
+
"has-dissent"
|
|
5987
|
+
if _is_dissent_downgraded(item, pbv)
|
|
5988
|
+
else _classify_plan_item_gate(item)
|
|
5989
|
+
)
|
|
5990
|
+
return {
|
|
5991
|
+
"id": item.get("id"),
|
|
5992
|
+
"classification": classification,
|
|
5993
|
+
"stateClassification": _state_classification(item, classification),
|
|
5994
|
+
"correctnessCritical": _is_correctness_critical(item),
|
|
5995
|
+
}
|
|
5996
|
+
|
|
5997
|
+
|
|
5933
5998
|
def plan_body_gate_summary(data: dict) -> dict | None:
|
|
5934
5999
|
"""The §5.5.9 gate as the round protocol's step 5 needs it — per-item
|
|
5935
6000
|
classification, the whole-gate value, and the `gateBlockedBy` causes, all
|
|
@@ -5951,13 +6016,7 @@ def plan_body_gate_summary(data: dict) -> dict | None:
|
|
|
5951
6016
|
if recomputed is None:
|
|
5952
6017
|
return None
|
|
5953
6018
|
items = [
|
|
5954
|
-
|
|
5955
|
-
"id": item.get("id"),
|
|
5956
|
-
"classification": "has-dissent"
|
|
5957
|
-
if _is_dissent_downgraded(item, pbv)
|
|
5958
|
-
else _classify_plan_item_gate(item),
|
|
5959
|
-
"correctnessCritical": _is_correctness_critical(item),
|
|
5960
|
-
}
|
|
6019
|
+
_gate_summary_item(item, pbv)
|
|
5961
6020
|
for item in (pbv.get("planItems") or [])
|
|
5962
6021
|
if isinstance(item, dict)
|
|
5963
6022
|
]
|
|
@@ -39,6 +39,12 @@ SECONDARY_BRIEF_FILENAME="validation-brief-secondary.md"
|
|
|
39
39
|
export OKSTRA_SKIP_INSTALL_CHECK="${OKSTRA_SKIP_INSTALL_CHECK:-1}"
|
|
40
40
|
export OKSTRA_CTL_SKIP_RECONCILE="${OKSTRA_CTL_SKIP_RECONCILE:-1}"
|
|
41
41
|
export OKSTRA_CTL_SKIP_BACKFILL="${OKSTRA_CTL_SKIP_BACKFILL:-1}"
|
|
42
|
+
# Same reason as the three above: the synthetic run must render the same way on
|
|
43
|
+
# every machine. cmux outranks tmux when present, so a maintainer running this
|
|
44
|
+
# from inside cmux would otherwise get the cmux adapter and a lead-session
|
|
45
|
+
# requirement this fixture never simulates. The cmux path has its own coverage
|
|
46
|
+
# in tests/run/test_cmux*.py and tests/contract/test_validate_session_conformance.py.
|
|
47
|
+
export CMUX_WORKSPACE_ID=""
|
|
42
48
|
|
|
43
49
|
# shellcheck source=lib/common.sh
|
|
44
50
|
source "$SCRIPT_DIR/lib/common.sh"
|