okstra 0.159.0 → 0.161.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture/storage-model.md +2 -0
  3. package/docs/architecture.md +2 -1
  4. package/docs/cli.md +8 -3
  5. package/docs/for-ai/README.md +2 -2
  6. package/docs/for-ai/skills/okstra-inspect.md +3 -0
  7. package/docs/for-ai/skills/okstra-run.md +2 -1
  8. package/docs/for-ai/skills/okstra-user-response.md +5 -5
  9. package/docs/project-structure-overview.md +5 -1
  10. package/docs/task-process/implementation.md +28 -0
  11. package/package.json +1 -1
  12. package/runtime/BUILD.json +2 -2
  13. package/runtime/bin/okstra-claude-exec.sh +4 -1
  14. package/runtime/prompts/host-orchestration/README.md +18 -0
  15. package/runtime/prompts/host-orchestration/implementation.md +57 -0
  16. package/runtime/prompts/launch.template.md +10 -1
  17. package/runtime/prompts/lead/adapters/claude-code.md +1 -1
  18. package/runtime/prompts/lead/adapters/cmux.md +67 -0
  19. package/runtime/prompts/lead/context-loader.md +5 -2
  20. package/runtime/prompts/lead/convergence.md +3 -1
  21. package/runtime/prompts/lead/plan-body-verification.md +21 -2
  22. package/runtime/prompts/lead/team-contract.md +2 -1
  23. package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
  24. package/runtime/prompts/profiles/_common-contract.md +3 -1
  25. package/runtime/prompts/profiles/implementation-planning.md +2 -0
  26. package/runtime/prompts/profiles/requirements-discovery.md +1 -1
  27. package/runtime/prompts/wizard/prompts.ko.json +3 -0
  28. package/runtime/python/okstra_ctl/clarification_items.py +9 -0
  29. package/runtime/python/okstra_ctl/cmux.py +531 -0
  30. package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
  31. package/runtime/python/okstra_ctl/convergence.py +168 -11
  32. package/runtime/python/okstra_ctl/dispatch_core.py +76 -7
  33. package/runtime/python/okstra_ctl/dispatch_state.py +16 -0
  34. package/runtime/python/okstra_ctl/error_issue.py +640 -0
  35. package/runtime/python/okstra_ctl/error_report.py +56 -0
  36. package/runtime/python/okstra_ctl/error_zip.py +23 -10
  37. package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
  38. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
  39. package/runtime/python/okstra_ctl/issue_signals.py +186 -0
  40. package/runtime/python/okstra_ctl/lead_runtime.py +30 -2
  41. package/runtime/python/okstra_ctl/paths.py +38 -0
  42. package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
  43. package/runtime/python/okstra_ctl/profile_show.py +134 -0
  44. package/runtime/python/okstra_ctl/recap.py +63 -0
  45. package/runtime/python/okstra_ctl/render.py +7 -2
  46. package/runtime/python/okstra_ctl/render_final_report.py +7 -22
  47. package/runtime/python/okstra_ctl/report_translation.py +4 -0
  48. package/runtime/python/okstra_ctl/report_views.py +7 -3
  49. package/runtime/python/okstra_ctl/run.py +54 -3
  50. package/runtime/python/okstra_ctl/run_audit.py +477 -0
  51. package/runtime/python/okstra_ctl/team.py +50 -11
  52. package/runtime/python/okstra_ctl/user_response.py +25 -10
  53. package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
  54. package/runtime/python/okstra_ctl/wizard.py +64 -10
  55. package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
  56. package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
  57. package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
  58. package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
  59. package/runtime/schemas/final-report-v1.0.schema.json +14 -0
  60. package/runtime/schemas/final-report-v2.0.schema.json +51 -1
  61. package/runtime/skills/okstra-inspect/SKILL.md +3 -1
  62. package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
  63. package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
  64. package/runtime/skills/okstra-run/SKILL.md +28 -10
  65. package/runtime/skills/okstra-user-response/SKILL.md +18 -18
  66. package/runtime/templates/reports/final-report.template.md +4 -0
  67. package/runtime/templates/reports/html/i18n/en.json +5 -1
  68. package/runtime/templates/reports/html/i18n/ko.json +5 -1
  69. package/runtime/templates/reports/html/macros/forms.html +15 -0
  70. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
  71. package/runtime/templates/reports/i18n/en.json +2 -0
  72. package/runtime/validators/validate-run.py +267 -208
  73. package/runtime/validators/validate-workflow.sh +6 -0
  74. package/runtime/validators/validate_session_conformance.py +135 -31
  75. package/src/cli-registry.mjs +34 -0
  76. package/src/commands/execute/incremental-scope.mjs +10 -0
  77. package/src/commands/execute/worker-audit-check.mjs +35 -0
  78. package/src/commands/inspect/error-issue.mjs +27 -0
  79. package/src/commands/inspect/profile-show.mjs +29 -0
  80. package/src/commands/inspect/run-audit.mjs +26 -0
@@ -66,6 +66,7 @@ from okstra_ctl.report_translation import ( # noqa: E402
66
66
  hangul_share,
67
67
  )
68
68
  from okstra_ctl.stage_citations import cited_stage_numbers # noqa: E402
69
+ from okstra_ctl.incremental_scope import coverage_row_blocked_on # noqa: E402
69
70
  from okstra_ctl.workflow import DEFAULT_NEXT_PHASE, PHASE_SEQUENCE # noqa: E402
70
71
  from okstra_ctl.md_table import ( # noqa: E402
71
72
  is_separator_row as _is_markdown_separator,
@@ -103,7 +104,10 @@ from okstra_ctl.worker_prompt_contract import ( # noqa: E402
103
104
  PromptRecord,
104
105
  validate_initial_prompt_records,
105
106
  )
106
- from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER # noqa: E402
107
+ from okstra_ctl.worker_audit_ledger import ( # noqa: E402
108
+ READING_CONFIRMATION_HEADING_RE,
109
+ check_worker_results_audit,
110
+ )
107
111
  from validate_analysis_report import validate_analysis_report # noqa: E402
108
112
  from okstra_ctl.convergence_engine import ( # noqa: E402
109
113
  grouped_input_digest,
@@ -1082,7 +1086,7 @@ def _validate_v2_ai_handoff(content: str, failures: list[str]) -> None:
1082
1086
  "schema-v2 AI handoff markdown contains human-only field "
1083
1087
  f"{field!r}; render it only in the task-specific HTML."
1084
1088
  )
1085
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1089
+ if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1086
1090
  failures.append(
1087
1091
  "final report contains a `## 0. Reading Confirmation` heading — "
1088
1092
  "Reading Confirmation lives in the worker audit sidecar, never "
@@ -1102,12 +1106,6 @@ _REPORT_INDEX_ANCHOR_RE = re.compile(r'<a id="report-index"')
1102
1106
  # never anchored — i.e. an un-anchored ID that the Index cannot link to.
1103
1107
  _UNANCHORED_ID_ROW_RE = re.compile(r"^\|[ \t]*\*{0,2}([A-Z]{1,4}-\d{3,})\b", re.MULTILINE)
1104
1108
 
1105
- # Reading Confirmation heading must NOT appear in the final-report — it
1106
- # belongs in the worker audit sidecar (`<worker>-audit-<task-type>-<seq>.md`).
1107
- _READING_CONFIRMATION_HEADING_RE = re.compile(
1108
- r"^##[ \t]+0\.[ \t]+Reading Confirmation\b", re.MULTILINE
1109
- )
1110
-
1111
1109
  # Empty Section 0 (Clarification Response Carried In) stub. When no
1112
1110
  # carry-in path is provided, the writer must OMIT the `## 0.` heading
1113
1111
  # entirely — emitting the heading followed by the "No prior clarification
@@ -2349,7 +2347,7 @@ def validate_report(
2349
2347
 
2350
2348
  # Reading Confirmation belongs in the worker audit sidecar, not the
2351
2349
  # user-facing final-report.
2352
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2350
+ if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2353
2351
  failures.append(
2354
2352
  "final report contains a `## 0. Reading Confirmation` heading — "
2355
2353
  "Reading Confirmation lives in the worker audit sidecar "
@@ -2385,27 +2383,6 @@ def validate_report(
2385
2383
  failures.append(f"final report contains {remedy}")
2386
2384
 
2387
2385
 
2388
- # Worker-results filename pattern: `<worker-role>-<task-type>-<seq>.md`.
2389
- # Every analysis-worker role name ends in `-worker` (`claude-worker`,
2390
- # `codex-worker`, `antigravity-worker`, `report-writer-worker`), so anchor the
2391
- # split on that suffix — otherwise `antigravity-worker-error-analysis-001.md`
2392
- # ambiguously parses as `worker=antigravity, task=worker-error-analysis`.
2393
- # Audit sidecars (`*-audit-*`) and errors sidecars (`.json`) are not
2394
- # matched here.
2395
- _WORKER_RESULT_BASENAME_RE = re.compile(
2396
- r"^(?P<worker>[a-z][a-z0-9-]*-worker)-(?P<task_type>[a-z][a-z-]*?)-(?P<seq>\d{3})\.md$"
2397
- )
2398
- _EVIDENCE_READ_RE = re.compile(
2399
- r"^- Evidence read: `(?P<path>[^`\n]+)`\s*$",
2400
- re.MULTILINE,
2401
- )
2402
- _FILE_LINE_CITATION_RE = re.compile(
2403
- r"`(?P<path>(?!https?://)[^`\n]+?):(?P<line>\d+(?:-\d+)?)`"
2404
- )
2405
- _EXTENSIONLESS_SOURCE_FILENAMES = frozenset(
2406
- {"Dockerfile", "Justfile", "Makefile", "Procfile", "Rakefile"}
2407
- )
2408
-
2409
2386
  _REPORT_BASENAME_SEQ_RE = re.compile(r"-(?P<seq>\d{3})(?:\.data)?\.(?:md|json)$")
2410
2387
 
2411
2388
 
@@ -2417,181 +2394,23 @@ def _report_run_seq(report_path: Path) -> str | None:
2417
2394
  return match.group("seq") if match else None
2418
2395
 
2419
2396
 
2420
- def _cited_file_paths(content: str) -> set[str]:
2421
- paths: set[str] = set()
2422
- for match in _FILE_LINE_CITATION_RE.finditer(content):
2423
- path = match.group("path")
2424
- if _looks_like_file_path(path):
2425
- paths.add(path)
2426
- return paths
2427
-
2428
-
2429
- def _looks_like_file_path(path: str) -> bool:
2430
- if (
2431
- not path
2432
- or path.startswith(("-", "$"))
2433
- or any(char.isspace() for char in path)
2434
- ):
2435
- return False
2436
- if re.fullmatch(r"[0-9a-fA-F]{7,64}", path):
2437
- return False
2438
- return (
2439
- "/" in path
2440
- or "." in Path(path).name
2441
- or Path(path).name in _EXTENSIONLESS_SOURCE_FILENAMES
2442
- )
2443
-
2444
-
2445
- def _audit_evidence_read_paths(content: str) -> set[str]:
2446
- return {
2447
- match.group("path")
2448
- for match in _EVIDENCE_READ_RE.finditer(content)
2449
- }
2450
-
2451
-
2452
- def _worker_prompt_path(
2453
- report_path: Path,
2454
- worker_role: str,
2455
- task_type: str,
2456
- seq: str,
2457
- ) -> Path:
2458
- return (
2459
- report_path.parent.parent
2460
- / "prompts"
2461
- / f"{worker_role}-prompt-{task_type}-{seq}.md"
2462
- )
2463
-
2464
-
2465
- def _validate_worker_evidence_read_ledger(
2466
- *,
2467
- report_path: Path,
2468
- worker_role: str,
2469
- task_type: str,
2470
- seq: str,
2471
- result_name: str,
2472
- result_content: str,
2473
- audit_path: Path,
2474
- failures: list[str],
2475
- ) -> None:
2476
- if worker_role == "report-writer-worker":
2477
- return
2478
- prompt_path = _worker_prompt_path(report_path, worker_role, task_type, seq)
2479
- try:
2480
- prompt_content = prompt_path.read_text(encoding="utf-8")
2481
- except OSError:
2482
- return
2483
- if EVIDENCE_LEDGER_HEADER not in prompt_content.splitlines():
2484
- return
2485
- try:
2486
- audit_content = audit_path.read_text(encoding="utf-8")
2487
- except OSError as exc:
2488
- failures.append(
2489
- f"worker audit sidecar unreadable: {audit_path.name} ({exc})"
2490
- )
2491
- return
2492
-
2493
- missing_paths = sorted(
2494
- _cited_file_paths(result_content) - _audit_evidence_read_paths(audit_content)
2495
- )
2496
- for missing_path in missing_paths:
2497
- failures.append(
2498
- f"worker `{worker_role}` result `{result_name}` cites "
2499
- f"`{missing_path}:line` without an Evidence read row for "
2500
- f"`{missing_path}` in `{audit_path.name}`"
2501
- )
2502
-
2503
-
2504
2397
  def validate_worker_results_audit(
2505
2398
  report_path: Path, task_type: str, failures: list[str]
2506
2399
  ) -> None:
2507
- """Enforce the worker audit sidecar contract.
2508
-
2509
- For every `worker-results/<worker>-<task-type>-<seq>.md` **this run**
2510
- produced (skipping the audit sidecar itself), the validator checks:
2511
-
2512
- 1. The main worker-results file does NOT contain a `## 0. Reading
2513
- Confirmation` heading. That block moved to the audit sidecar with
2514
- the report-format readability pass.
2515
- 2. The matching audit sidecar exists at
2516
- `<worker>-audit-<task-type>-<seq>.md`. Missing sidecar means the
2517
- worker silently skipped the reading-confirmation step.
2518
- 3. For new prompts carrying the required-v1 marker, every canonical
2519
- backticked `path:line` citation has a matching Evidence read row in
2520
- that audit sidecar. Historical prompts without the marker retain the
2521
- existence-only contract.
2522
-
2523
- Scoped to this run's seq. `worker-results/` accumulates every run's
2524
- artifacts, so scanning the whole directory judged a run by files it did
2525
- not produce: a task that once opted a worker in and later dropped it from
2526
- the roster failed forever on that worker's old sidecar, with no legitimate
2527
- remedy — the lead can neither fabricate a sidecar for a worker it never
2528
- dispatched nor delete a prior run's audit record.
2529
- """
2530
- # `report_path` is `runs/<task-type>/reports/final-report-...md`; the
2531
- # sibling `worker-results/` directory holds every worker artifact.
2532
- worker_results_dir = report_path.parent.parent / "worker-results"
2533
- if not worker_results_dir.is_dir():
2534
- # No worker-results directory means no analysis workers ran (e.g.
2535
- # `release-handoff` which is single-lead). Nothing to enforce.
2536
- return
2400
+ """Enforce the worker audit sidecar contract at Phase 7.
2537
2401
 
2538
- run_seq = _report_run_seq(report_path)
2539
-
2540
- for path in sorted(worker_results_dir.glob("*.md")):
2541
- name = path.name
2542
- if "-audit-" in name:
2543
- continue
2544
- match = _WORKER_RESULT_BASENAME_RE.match(name)
2545
- if match is None:
2546
- # Files that don't match the canonical pattern (e.g. ad-hoc
2547
- # notes left by the operator) are out of contract scope.
2548
- continue
2549
- if match.group("task_type") != task_type:
2550
- # Cross-phase artifacts shouldn't appear here; skip rather
2551
- # than fail to keep the check focused on the current phase.
2552
- continue
2553
- if run_seq is not None and match.group("seq") != run_seq:
2554
- # A prior run's artifact. Its contract was judged when it ran.
2555
- continue
2556
-
2557
- worker_role = match.group("worker")
2558
- seq = match.group("seq")
2559
- rel = path.name
2560
- try:
2561
- content = path.read_text()
2562
- except OSError as exc:
2563
- failures.append(f"worker-results file unreadable: {rel} ({exc})")
2564
- continue
2565
-
2566
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2567
- failures.append(
2568
- f"worker-results file `{rel}` contains a `## 0. Reading "
2569
- f"Confirmation` heading — that block moved to the audit "
2570
- f"sidecar (`{worker_role}-audit-{task_type}-{seq}.md`). "
2571
- f"Remove the §0 heading + body from the main file and "
2572
- f"write a fresh sidecar."
2573
- )
2574
-
2575
- audit_path = worker_results_dir / f"{worker_role}-audit-{task_type}-{seq}.md"
2576
- if not audit_path.exists():
2577
- failures.append(
2578
- f"worker `{worker_role}` produced `{rel}` but no audit sidecar "
2579
- f"at `{audit_path.name}` — the sidecar must carry the Reading "
2580
- f"Confirmation block (one short line per input file). Workers "
2581
- f"write this in the same step as the main worker-results file."
2582
- )
2583
- continue
2584
-
2585
- _validate_worker_evidence_read_ledger(
2586
- report_path=report_path,
2587
- worker_role=worker_role,
2588
- task_type=task_type,
2589
- seq=seq,
2590
- result_name=rel,
2591
- result_content=content,
2592
- audit_path=audit_path,
2593
- failures=failures,
2594
- )
2402
+ The rules themselves live in `okstra_ctl.worker_audit_ledger` so that
2403
+ `okstra worker-audit-check` can apply the identical checks mid-run, while
2404
+ the worker session is still alive and can fix its own citations. All this
2405
+ wrapper adds is the Phase 7 anchor: the run directory and this run's seq,
2406
+ both read off the report path.
2407
+ """
2408
+ failures.extend(check_worker_results_audit(
2409
+ # `report_path` is `runs/<task-type>/reports/final-report-...md`.
2410
+ report_path.parent.parent,
2411
+ task_type,
2412
+ _report_run_seq(report_path),
2413
+ ))
2595
2414
 
2596
2415
 
2597
2416
  def validate_team_state_usage(team_state: dict, failures: list[str]) -> None:
@@ -3307,6 +3126,9 @@ def validate_final_report_data(
3307
3126
  _validate_verdict_card_fields(data, failures)
3308
3127
  # Phase-agnostic: the coverage critic runs in every finding-producing phase.
3309
3128
  _validate_unverified_critic_gaps_recorded(data, failures)
3129
+ # Called here rather than from a task-type branch: four profiles raise
3130
+ # clarification rows, and the gate scopes itself by task type internally.
3131
+ _validate_clarification_options(data, failures)
3310
3132
 
3311
3133
  task_type = (data.get("header") or {}).get("taskType")
3312
3134
  _validate_verifier_fail_blocks_verdict(data, failures)
@@ -3331,6 +3153,8 @@ def validate_final_report_data(
3331
3153
  print(f"validate-run: warning: {warning}", file=sys.stderr)
3332
3154
  _validate_supersession_ledger(data, failures)
3333
3155
  _validate_clarification_evidence_note(data, failures)
3156
+ _validate_approval_clarification_backtrace(data, failures)
3157
+ _validate_rerun_guidance(data, failures)
3334
3158
  _validate_variation_point_analysis(
3335
3159
  (data.get("implementationPlanning") or {}).get("variationPointAnalysis"),
3336
3160
  resolve_architecture(_project_root_from_report(report_path)),
@@ -3658,6 +3482,38 @@ def _self_fix_budget_exhausted(pbv: dict) -> bool:
3658
3482
  )
3659
3483
 
3660
3484
 
3485
+ def _state_classification(item: dict, gate_class: str) -> str:
3486
+ """This item's `planItems[].rounds[].classification` for the state file.
3487
+
3488
+ The gate classifier deliberately folds `partial-consensus` and
3489
+ `dissent-isolated` into `has-dissent` — only the majority-disagree boundary
3490
+ moves the gate. The state file records the finer label, and the information
3491
+ to recover it is in the same verdicts, so the mapping lives beside the
3492
+ classifier rather than being re-invented by each lead.
3493
+
3494
+ *gate_class* is passed in rather than recomputed so that the caller's
3495
+ effective classification — which may have been downgraded by
3496
+ `_is_dissent_downgraded` — is the one this translates.
3497
+
3498
+ `contested` never appears: it is only meaningful at `maxRounds > 1`, and at
3499
+ the default `maxRounds=1` the round protocol folds any otherwise-unresolved
3500
+ item into `partial-consensus`.
3501
+ """
3502
+ if gate_class == "all-non-result":
3503
+ # No non-error vote at all is the `needs-reverify` shape taken to its
3504
+ # limit — "fewer than 2 participating votes" covers zero.
3505
+ return "needs-reverify"
3506
+ if gate_class != "has-dissent":
3507
+ return gate_class
3508
+ dissenting = sum(
3509
+ 1
3510
+ for vote in (item.get("verdicts") or [])
3511
+ if isinstance(vote, dict)
3512
+ and str(vote.get("verdict") or "").strip().upper() == "DISAGREE"
3513
+ )
3514
+ return "dissent-isolated" if dissenting == 1 else "partial-consensus"
3515
+
3516
+
3661
3517
  def _is_dissent_downgraded(item: dict, pbv: dict) -> bool:
3662
3518
  """Whether a surviving `majority-disagree` item stops blocking approval.
3663
3519
 
@@ -4180,6 +4036,200 @@ def _validate_clarification_evidence_note(data: dict, failures: list[str]) -> No
4180
4036
  )
4181
4037
 
4182
4038
 
4039
+ _CLARIFICATION_OPTION_SCHEMA_VERSION = "2.0"
4040
+ # The four profiles that read `_clarification-recommendation.md`. Unlike the
4041
+ # evidence-note gate above — which is called from inside the
4042
+ # `implementation-planning` branch — this one is called phase-agnostically, so
4043
+ # the task-type filter here is the only thing scoping it.
4044
+ _CLARIFICATION_OPTION_TASK_TYPES = frozenset({
4045
+ "error-analysis",
4046
+ "implementation-planning",
4047
+ "improvement-discovery",
4048
+ "requirements-discovery",
4049
+ })
4050
+ # `in-repo` and `cross-repo` answer the same question, so exactly one may hold.
4051
+ _REACH_TOKENS = frozenset({"in-repo", "cross-repo"})
4052
+ _LEGACY_EXPECTED_FORM_RE = re.compile(r"\b(?:Recommended|Alternatives):")
4053
+
4054
+
4055
+ def _validate_clarification_options(data: dict, failures: list[str]) -> None:
4056
+ """A `decision` row must carry its choices as data, not as prose.
4057
+
4058
+ The choices used to live inside the `expectedForm` string, where two
4059
+ separate parsers split them differently and neither was checked — the board
4060
+ the user picked from could disagree with the board the report meant.
4061
+ Structured options remove the parsing; this gate keeps the structure
4062
+ honest. Whether an impact claim is *true* is the adversarial round's job,
4063
+ exactly as with `Evidence checked:`.
4064
+
4065
+ schema-v1 is exempt because it cannot comply: it keeps clarifications as a
4066
+ Markdown table of strings and its schema forbids an `options` property, so
4067
+ demanding one would fail every v1 `decision` row for a structure the format
4068
+ has no place to hold.
4069
+ """
4070
+ if data.get("schemaVersion") != _CLARIFICATION_OPTION_SCHEMA_VERSION:
4071
+ return
4072
+ task_type = (data.get("header") or {}).get("taskType")
4073
+ if task_type not in _CLARIFICATION_OPTION_TASK_TYPES:
4074
+ return
4075
+ for row in data.get("clarificationItems") or []:
4076
+ if not isinstance(row, dict) or row.get("kind") != "decision":
4077
+ continue
4078
+ row_id = str(row.get("id") or "<unknown>")
4079
+ _validate_option_set(row.get("options"), row_id, failures)
4080
+ if _LEGACY_EXPECTED_FORM_RE.search(str(row.get("expectedForm") or "")):
4081
+ failures.append(
4082
+ f"final-report data.json: clarification `{row_id}` still encodes "
4083
+ "its choices in `expectedForm` (`Recommended:` / `Alternatives:`). "
4084
+ "Choices belong in `options[]`; `expectedForm` states only the "
4085
+ "shape of the answer. Two sources for one fact leave consumers "
4086
+ "disagreeing about which is authoritative."
4087
+ )
4088
+
4089
+
4090
+ def _validate_option_set(
4091
+ options: object, row_id: str, failures: list[str]
4092
+ ) -> None:
4093
+ """Check one `decision` row's `options[]` for pickability and reach."""
4094
+ if not isinstance(options, list) or len(options) < 2:
4095
+ failures.append(
4096
+ f"final-report data.json: clarification `{row_id}` is a `decision` "
4097
+ "but does not offer at least two `options[]`. A decision the user "
4098
+ "cannot choose between is not a decision."
4099
+ )
4100
+ return
4101
+ entries = [option for option in options if isinstance(option, dict)]
4102
+ recommended = sum(1 for option in entries if option.get("role") == "recommended")
4103
+ if recommended != 1:
4104
+ failures.append(
4105
+ f"final-report data.json: clarification `{row_id}` must carry exactly "
4106
+ f"one `role: recommended` option (found {recommended}). The reader "
4107
+ "needs to know which answer the run stands behind."
4108
+ )
4109
+ for index, option in enumerate(entries):
4110
+ tokens = option.get("scopeImpact")
4111
+ reach = (
4112
+ [token for token in tokens if token in _REACH_TOKENS]
4113
+ if isinstance(tokens, list)
4114
+ else []
4115
+ )
4116
+ if len(reach) != 1:
4117
+ failures.append(
4118
+ f"final-report data.json: clarification `{row_id}` option "
4119
+ f"[{index}] must declare exactly one reach token — `in-repo` or "
4120
+ f"`cross-repo` (found {len(reach)}). The reader cannot weigh an "
4121
+ "option whose reach is unstated or self-contradictory."
4122
+ )
4123
+
4124
+
4125
+ def _has_clarification_backtrace(
4126
+ row_id: str, plan_items: object, coverage: object
4127
+ ) -> bool:
4128
+ """Whether the plan records anything this clarification blocks.
4129
+
4130
+ Two link shapes, both authored by the same run: the `P-*` plan item that
4131
+ carries the `clarificationId`, and the requirement-coverage row blocked on
4132
+ the id. `incremental-scope` resolves impacted stages from exactly these
4133
+ two, and the coverage side goes through its predicate so the gate and the
4134
+ resolver cannot disagree about what counts as a link.
4135
+ """
4136
+ if isinstance(plan_items, list) and any(
4137
+ isinstance(item, dict) and item.get("clarificationId") == row_id
4138
+ for item in plan_items
4139
+ ):
4140
+ return True
4141
+ return isinstance(coverage, list) and any(
4142
+ coverage_row_blocked_on(row, row_id) for row in coverage
4143
+ )
4144
+
4145
+
4146
+ def _validate_approval_clarification_backtrace(
4147
+ data: dict, failures: list[str]
4148
+ ) -> None:
4149
+ """An approval blocker must record what it blocks.
4150
+
4151
+ `_validate_plan_body_clarification_matching` already walks the other
4152
+ direction — a majority-disagree plan item must cite a `blocks: approval`
4153
+ row. Nothing walked this way, so a row could withhold approval while
4154
+ recording no blast radius at all. The cost lands on the re-run:
4155
+ `incremental-scope` resolves impacted stages from these links and treats an
4156
+ id that traces to no stage as grounds to re-verify everything, so one
4157
+ unlinked blocker turns a narrow re-run into a full one.
4158
+ """
4159
+ if (data.get("header") or {}).get("taskType") != "implementation-planning":
4160
+ return
4161
+ planning = data.get("implementationPlanning")
4162
+ if not isinstance(planning, dict):
4163
+ return
4164
+ coverage = planning.get("requirementCoverage")
4165
+ verification = planning.get("planBodyVerification")
4166
+ plan_items = (
4167
+ verification.get("planItems") if isinstance(verification, dict) else None
4168
+ )
4169
+ for row in data.get("clarificationItems") or []:
4170
+ if not isinstance(row, dict) or row.get("blocks") != "approval":
4171
+ continue
4172
+ row_id = str(row.get("id") or "<unknown>")
4173
+ if _has_clarification_backtrace(row_id, plan_items, coverage):
4174
+ continue
4175
+ failures.append(
4176
+ f"final-report data.json: clarification `{row_id}` blocks approval "
4177
+ "but has no back-trace into the plan — no plan item carries it as "
4178
+ "`clarificationId`, and no requirement-coverage row is `blocked "
4179
+ f"{row_id}` in its `status` or `approvalDisposition`. An item that "
4180
+ "withholds approval without recording what it affects forces the "
4181
+ "next re-run to re-verify everything."
4182
+ )
4183
+
4184
+
4185
+ _RERUN_FLAG = "--answered-clarifications"
4186
+
4187
+
4188
+ def _next_step_texts(steps: object) -> list[str]:
4189
+ """Every reader-visible string in `recommendedNextSteps`, prose and command."""
4190
+ texts: list[str] = []
4191
+ for step in steps if isinstance(steps, list) else []:
4192
+ if not isinstance(step, dict):
4193
+ continue
4194
+ texts.append(str(step.get("text") or ""))
4195
+ for command in step.get("commands") or []:
4196
+ if isinstance(command, dict):
4197
+ texts.append(str(command.get("claudeCode") or ""))
4198
+ texts.append(str(command.get("terminal") or ""))
4199
+ return texts
4200
+
4201
+
4202
+ def _validate_rerun_guidance(data: dict, failures: list[str]) -> None:
4203
+ """A report that withholds approval must say how to come back from it.
4204
+
4205
+ The way forward — answer the blockers, then re-run carrying those ids —
4206
+ lived only in the lead prompt, which is read *after* the next run has
4207
+ already started. The person who has to act reads the report instead, and it
4208
+ told them nothing about the next command. Requiring the flag by name is a
4209
+ low bar deliberately: it does not check that the rest of the step is right,
4210
+ only that the report stops leaving the reader to work the mechanics out.
4211
+ """
4212
+ if (data.get("header") or {}).get("taskType") != "implementation-planning":
4213
+ return
4214
+ has_blocker = any(
4215
+ isinstance(row, dict) and row.get("blocks") == "approval"
4216
+ for row in data.get("clarificationItems") or []
4217
+ )
4218
+ if not has_blocker:
4219
+ return
4220
+ if any(_RERUN_FLAG in text for text in _next_step_texts(
4221
+ data.get("recommendedNextSteps")
4222
+ )):
4223
+ return
4224
+ failures.append(
4225
+ "final-report data.json: this plan withholds approval on a "
4226
+ "`blocks: approval` clarification, but no `recommendedNextSteps` entry "
4227
+ f"tells the reader how to resume — name the `{_RERUN_FLAG}` re-run in a "
4228
+ "step's `text` or one of its `commands`. `okstra recap assemble` "
4229
+ "prints the exact ids and flag value once the answers are recorded."
4230
+ )
4231
+
4232
+
4183
4233
  def _validate_self_fix_grouping(data: dict, failures: list[str]) -> None:
4184
4234
  """A self-fix round must be instructed by cause, not as a flat item list.
4185
4235
 
@@ -5930,6 +5980,21 @@ def validate_plan_body_section(
5930
5980
  return [*_detect_self_fix_recurrence(pbv), *_detect_uniform_verifier(pbv)]
5931
5981
 
5932
5982
 
5983
+ def _gate_summary_item(item: dict, pbv: dict) -> dict:
5984
+ """One `gate.items[]` row: the gate class plus its state-file counterpart."""
5985
+ classification = (
5986
+ "has-dissent"
5987
+ if _is_dissent_downgraded(item, pbv)
5988
+ else _classify_plan_item_gate(item)
5989
+ )
5990
+ return {
5991
+ "id": item.get("id"),
5992
+ "classification": classification,
5993
+ "stateClassification": _state_classification(item, classification),
5994
+ "correctnessCritical": _is_correctness_critical(item),
5995
+ }
5996
+
5997
+
5933
5998
  def plan_body_gate_summary(data: dict) -> dict | None:
5934
5999
  """The §5.5.9 gate as the round protocol's step 5 needs it — per-item
5935
6000
  classification, the whole-gate value, and the `gateBlockedBy` causes, all
@@ -5951,13 +6016,7 @@ def plan_body_gate_summary(data: dict) -> dict | None:
5951
6016
  if recomputed is None:
5952
6017
  return None
5953
6018
  items = [
5954
- {
5955
- "id": item.get("id"),
5956
- "classification": "has-dissent"
5957
- if _is_dissent_downgraded(item, pbv)
5958
- else _classify_plan_item_gate(item),
5959
- "correctnessCritical": _is_correctness_critical(item),
5960
- }
6019
+ _gate_summary_item(item, pbv)
5961
6020
  for item in (pbv.get("planItems") or [])
5962
6021
  if isinstance(item, dict)
5963
6022
  ]
@@ -39,6 +39,12 @@ SECONDARY_BRIEF_FILENAME="validation-brief-secondary.md"
39
39
  export OKSTRA_SKIP_INSTALL_CHECK="${OKSTRA_SKIP_INSTALL_CHECK:-1}"
40
40
  export OKSTRA_CTL_SKIP_RECONCILE="${OKSTRA_CTL_SKIP_RECONCILE:-1}"
41
41
  export OKSTRA_CTL_SKIP_BACKFILL="${OKSTRA_CTL_SKIP_BACKFILL:-1}"
42
+ # Same reason as the three above: the synthetic run must render the same way on
43
+ # every machine. cmux outranks tmux when present, so a maintainer running this
44
+ # from inside cmux would otherwise get the cmux adapter and a lead-session
45
+ # requirement this fixture never simulates. The cmux path has its own coverage
46
+ # in tests/run/test_cmux*.py and tests/contract/test_validate_session_conformance.py.
47
+ export CMUX_WORKSPACE_ID=""
42
48
 
43
49
  # shellcheck source=lib/common.sh
44
50
  source "$SCRIPT_DIR/lib/common.sh"