okstra 0.158.1 → 0.160.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture/storage-model.md +2 -0
  3. package/docs/architecture.md +1 -1
  4. package/docs/cli.md +8 -3
  5. package/docs/for-ai/README.md +2 -2
  6. package/docs/for-ai/skills/okstra-inspect.md +3 -0
  7. package/docs/for-ai/skills/okstra-run.md +2 -1
  8. package/docs/for-ai/skills/okstra-user-response.md +5 -5
  9. package/docs/project-structure-overview.md +5 -1
  10. package/docs/task-process/implementation.md +28 -0
  11. package/package.json +1 -1
  12. package/runtime/BUILD.json +2 -2
  13. package/runtime/agents/workers/report-writer-worker.md +1 -1
  14. package/runtime/bin/okstra-claude-exec.sh +4 -1
  15. package/runtime/prompts/host-orchestration/README.md +18 -0
  16. package/runtime/prompts/host-orchestration/implementation.md +57 -0
  17. package/runtime/prompts/launch.template.md +10 -1
  18. package/runtime/prompts/lead/adapters/claude-code.md +1 -1
  19. package/runtime/prompts/lead/context-loader.md +5 -2
  20. package/runtime/prompts/lead/convergence.md +3 -1
  21. package/runtime/prompts/lead/plan-body-verification.md +21 -2
  22. package/runtime/prompts/lead/report-writer.md +1 -1
  23. package/runtime/prompts/lead/team-contract.md +2 -1
  24. package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
  25. package/runtime/prompts/profiles/_common-contract.md +3 -1
  26. package/runtime/prompts/profiles/implementation-planning.md +2 -0
  27. package/runtime/prompts/profiles/requirements-discovery.md +1 -1
  28. package/runtime/prompts/wizard/prompts.ko.json +3 -0
  29. package/runtime/python/okstra_ctl/clarification_items.py +9 -0
  30. package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
  31. package/runtime/python/okstra_ctl/convergence.py +168 -11
  32. package/runtime/python/okstra_ctl/dispatch_core.py +4 -2
  33. package/runtime/python/okstra_ctl/error_issue.py +640 -0
  34. package/runtime/python/okstra_ctl/error_report.py +56 -0
  35. package/runtime/python/okstra_ctl/error_zip.py +23 -10
  36. package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
  37. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
  38. package/runtime/python/okstra_ctl/issue_signals.py +186 -0
  39. package/runtime/python/okstra_ctl/paths.py +38 -0
  40. package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
  41. package/runtime/python/okstra_ctl/profile_show.py +134 -0
  42. package/runtime/python/okstra_ctl/recap.py +63 -0
  43. package/runtime/python/okstra_ctl/render_final_report.py +11 -62
  44. package/runtime/python/okstra_ctl/report_html/filters.py +6 -1
  45. package/runtime/python/okstra_ctl/report_html/render.py +9 -8
  46. package/runtime/python/okstra_ctl/report_html/run_usage.py +110 -0
  47. package/runtime/python/okstra_ctl/report_html/view_models/error_analysis.py +69 -16
  48. package/runtime/python/okstra_ctl/report_html/visualizations.py +107 -14
  49. package/runtime/python/okstra_ctl/report_translation.py +4 -0
  50. package/runtime/python/okstra_ctl/report_views.py +7 -3
  51. package/runtime/python/okstra_ctl/run.py +41 -2
  52. package/runtime/python/okstra_ctl/run_audit.py +477 -0
  53. package/runtime/python/okstra_ctl/usage_cells.py +47 -0
  54. package/runtime/python/okstra_ctl/user_response.py +25 -10
  55. package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
  56. package/runtime/python/okstra_ctl/wizard.py +64 -10
  57. package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
  58. package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
  59. package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
  60. package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
  61. package/runtime/schemas/final-report-v1.0.schema.json +14 -0
  62. package/runtime/schemas/final-report-v2.0.schema.json +56 -2
  63. package/runtime/skills/okstra-inspect/SKILL.md +3 -1
  64. package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
  65. package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
  66. package/runtime/skills/okstra-run/SKILL.md +28 -10
  67. package/runtime/skills/okstra-user-response/SKILL.md +18 -18
  68. package/runtime/templates/reports/final-report.template.md +4 -0
  69. package/runtime/templates/reports/html/assets/base.css +14 -1
  70. package/runtime/templates/reports/html/base.template.html +42 -0
  71. package/runtime/templates/reports/html/i18n/en.json +30 -1
  72. package/runtime/templates/reports/html/i18n/ko.json +30 -1
  73. package/runtime/templates/reports/html/macros/forms.html +15 -0
  74. package/runtime/templates/reports/html/macros/visualizations.html +3 -2
  75. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
  76. package/runtime/templates/reports/i18n/en.json +2 -0
  77. package/runtime/validators/validate-run.py +331 -208
  78. package/runtime/validators/validate_session_conformance.py +102 -32
  79. package/src/cli-registry.mjs +34 -0
  80. package/src/commands/execute/incremental-scope.mjs +10 -0
  81. package/src/commands/execute/worker-audit-check.mjs +35 -0
  82. package/src/commands/inspect/error-issue.mjs +27 -0
  83. package/src/commands/inspect/profile-show.mjs +29 -0
  84. package/src/commands/inspect/run-audit.mjs +26 -0
@@ -66,6 +66,7 @@ from okstra_ctl.report_translation import ( # noqa: E402
66
66
  hangul_share,
67
67
  )
68
68
  from okstra_ctl.stage_citations import cited_stage_numbers # noqa: E402
69
+ from okstra_ctl.incremental_scope import coverage_row_blocked_on # noqa: E402
69
70
  from okstra_ctl.workflow import DEFAULT_NEXT_PHASE, PHASE_SEQUENCE # noqa: E402
70
71
  from okstra_ctl.md_table import ( # noqa: E402
71
72
  is_separator_row as _is_markdown_separator,
@@ -103,7 +104,10 @@ from okstra_ctl.worker_prompt_contract import ( # noqa: E402
103
104
  PromptRecord,
104
105
  validate_initial_prompt_records,
105
106
  )
106
- from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER # noqa: E402
107
+ from okstra_ctl.worker_audit_ledger import ( # noqa: E402
108
+ READING_CONFIRMATION_HEADING_RE,
109
+ check_worker_results_audit,
110
+ )
107
111
  from validate_analysis_report import validate_analysis_report # noqa: E402
108
112
  from okstra_ctl.convergence_engine import ( # noqa: E402
109
113
  grouped_input_digest,
@@ -1082,7 +1086,7 @@ def _validate_v2_ai_handoff(content: str, failures: list[str]) -> None:
1082
1086
  "schema-v2 AI handoff markdown contains human-only field "
1083
1087
  f"{field!r}; render it only in the task-specific HTML."
1084
1088
  )
1085
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1089
+ if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
1086
1090
  failures.append(
1087
1091
  "final report contains a `## 0. Reading Confirmation` heading — "
1088
1092
  "Reading Confirmation lives in the worker audit sidecar, never "
@@ -1102,12 +1106,6 @@ _REPORT_INDEX_ANCHOR_RE = re.compile(r'<a id="report-index"')
1102
1106
  # never anchored — i.e. an un-anchored ID that the Index cannot link to.
1103
1107
  _UNANCHORED_ID_ROW_RE = re.compile(r"^\|[ \t]*\*{0,2}([A-Z]{1,4}-\d{3,})\b", re.MULTILINE)
1104
1108
 
1105
- # Reading Confirmation heading must NOT appear in the final-report — it
1106
- # belongs in the worker audit sidecar (`<worker>-audit-<task-type>-<seq>.md`).
1107
- _READING_CONFIRMATION_HEADING_RE = re.compile(
1108
- r"^##[ \t]+0\.[ \t]+Reading Confirmation\b", re.MULTILINE
1109
- )
1110
-
1111
1109
  # Empty Section 0 (Clarification Response Carried In) stub. When no
1112
1110
  # carry-in path is provided, the writer must OMIT the `## 0.` heading
1113
1111
  # entirely — emitting the heading followed by the "No prior clarification
@@ -2349,7 +2347,7 @@ def validate_report(
2349
2347
 
2350
2348
  # Reading Confirmation belongs in the worker audit sidecar, not the
2351
2349
  # user-facing final-report.
2352
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2350
+ if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2353
2351
  failures.append(
2354
2352
  "final report contains a `## 0. Reading Confirmation` heading — "
2355
2353
  "Reading Confirmation lives in the worker audit sidecar "
@@ -2385,27 +2383,6 @@ def validate_report(
2385
2383
  failures.append(f"final report contains {remedy}")
2386
2384
 
2387
2385
 
2388
- # Worker-results filename pattern: `<worker-role>-<task-type>-<seq>.md`.
2389
- # Every analysis-worker role name ends in `-worker` (`claude-worker`,
2390
- # `codex-worker`, `antigravity-worker`, `report-writer-worker`), so anchor the
2391
- # split on that suffix — otherwise `antigravity-worker-error-analysis-001.md`
2392
- # ambiguously parses as `worker=antigravity, task=worker-error-analysis`.
2393
- # Audit sidecars (`*-audit-*`) and errors sidecars (`.json`) are not
2394
- # matched here.
2395
- _WORKER_RESULT_BASENAME_RE = re.compile(
2396
- r"^(?P<worker>[a-z][a-z0-9-]*-worker)-(?P<task_type>[a-z][a-z-]*?)-(?P<seq>\d{3})\.md$"
2397
- )
2398
- _EVIDENCE_READ_RE = re.compile(
2399
- r"^- Evidence read: `(?P<path>[^`\n]+)`\s*$",
2400
- re.MULTILINE,
2401
- )
2402
- _FILE_LINE_CITATION_RE = re.compile(
2403
- r"`(?P<path>(?!https?://)[^`\n]+?):(?P<line>\d+(?:-\d+)?)`"
2404
- )
2405
- _EXTENSIONLESS_SOURCE_FILENAMES = frozenset(
2406
- {"Dockerfile", "Justfile", "Makefile", "Procfile", "Rakefile"}
2407
- )
2408
-
2409
2386
  _REPORT_BASENAME_SEQ_RE = re.compile(r"-(?P<seq>\d{3})(?:\.data)?\.(?:md|json)$")
2410
2387
 
2411
2388
 
@@ -2417,181 +2394,23 @@ def _report_run_seq(report_path: Path) -> str | None:
2417
2394
  return match.group("seq") if match else None
2418
2395
 
2419
2396
 
2420
- def _cited_file_paths(content: str) -> set[str]:
2421
- paths: set[str] = set()
2422
- for match in _FILE_LINE_CITATION_RE.finditer(content):
2423
- path = match.group("path")
2424
- if _looks_like_file_path(path):
2425
- paths.add(path)
2426
- return paths
2427
-
2428
-
2429
- def _looks_like_file_path(path: str) -> bool:
2430
- if (
2431
- not path
2432
- or path.startswith(("-", "$"))
2433
- or any(char.isspace() for char in path)
2434
- ):
2435
- return False
2436
- if re.fullmatch(r"[0-9a-fA-F]{7,64}", path):
2437
- return False
2438
- return (
2439
- "/" in path
2440
- or "." in Path(path).name
2441
- or Path(path).name in _EXTENSIONLESS_SOURCE_FILENAMES
2442
- )
2443
-
2444
-
2445
- def _audit_evidence_read_paths(content: str) -> set[str]:
2446
- return {
2447
- match.group("path")
2448
- for match in _EVIDENCE_READ_RE.finditer(content)
2449
- }
2450
-
2451
-
2452
- def _worker_prompt_path(
2453
- report_path: Path,
2454
- worker_role: str,
2455
- task_type: str,
2456
- seq: str,
2457
- ) -> Path:
2458
- return (
2459
- report_path.parent.parent
2460
- / "prompts"
2461
- / f"{worker_role}-prompt-{task_type}-{seq}.md"
2462
- )
2463
-
2464
-
2465
- def _validate_worker_evidence_read_ledger(
2466
- *,
2467
- report_path: Path,
2468
- worker_role: str,
2469
- task_type: str,
2470
- seq: str,
2471
- result_name: str,
2472
- result_content: str,
2473
- audit_path: Path,
2474
- failures: list[str],
2475
- ) -> None:
2476
- if worker_role == "report-writer-worker":
2477
- return
2478
- prompt_path = _worker_prompt_path(report_path, worker_role, task_type, seq)
2479
- try:
2480
- prompt_content = prompt_path.read_text(encoding="utf-8")
2481
- except OSError:
2482
- return
2483
- if EVIDENCE_LEDGER_HEADER not in prompt_content.splitlines():
2484
- return
2485
- try:
2486
- audit_content = audit_path.read_text(encoding="utf-8")
2487
- except OSError as exc:
2488
- failures.append(
2489
- f"worker audit sidecar unreadable: {audit_path.name} ({exc})"
2490
- )
2491
- return
2492
-
2493
- missing_paths = sorted(
2494
- _cited_file_paths(result_content) - _audit_evidence_read_paths(audit_content)
2495
- )
2496
- for missing_path in missing_paths:
2497
- failures.append(
2498
- f"worker `{worker_role}` result `{result_name}` cites "
2499
- f"`{missing_path}:line` without an Evidence read row for "
2500
- f"`{missing_path}` in `{audit_path.name}`"
2501
- )
2502
-
2503
-
2504
2397
  def validate_worker_results_audit(
2505
2398
  report_path: Path, task_type: str, failures: list[str]
2506
2399
  ) -> None:
2507
- """Enforce the worker audit sidecar contract.
2508
-
2509
- For every `worker-results/<worker>-<task-type>-<seq>.md` **this run**
2510
- produced (skipping the audit sidecar itself), the validator checks:
2511
-
2512
- 1. The main worker-results file does NOT contain a `## 0. Reading
2513
- Confirmation` heading. That block moved to the audit sidecar with
2514
- the report-format readability pass.
2515
- 2. The matching audit sidecar exists at
2516
- `<worker>-audit-<task-type>-<seq>.md`. Missing sidecar means the
2517
- worker silently skipped the reading-confirmation step.
2518
- 3. For new prompts carrying the required-v1 marker, every canonical
2519
- backticked `path:line` citation has a matching Evidence read row in
2520
- that audit sidecar. Historical prompts without the marker retain the
2521
- existence-only contract.
2522
-
2523
- Scoped to this run's seq. `worker-results/` accumulates every run's
2524
- artifacts, so scanning the whole directory judged a run by files it did
2525
- not produce: a task that once opted a worker in and later dropped it from
2526
- the roster failed forever on that worker's old sidecar, with no legitimate
2527
- remedy — the lead can neither fabricate a sidecar for a worker it never
2528
- dispatched nor delete a prior run's audit record.
2529
- """
2530
- # `report_path` is `runs/<task-type>/reports/final-report-...md`; the
2531
- # sibling `worker-results/` directory holds every worker artifact.
2532
- worker_results_dir = report_path.parent.parent / "worker-results"
2533
- if not worker_results_dir.is_dir():
2534
- # No worker-results directory means no analysis workers ran (e.g.
2535
- # `release-handoff` which is single-lead). Nothing to enforce.
2536
- return
2537
-
2538
- run_seq = _report_run_seq(report_path)
2539
-
2540
- for path in sorted(worker_results_dir.glob("*.md")):
2541
- name = path.name
2542
- if "-audit-" in name:
2543
- continue
2544
- match = _WORKER_RESULT_BASENAME_RE.match(name)
2545
- if match is None:
2546
- # Files that don't match the canonical pattern (e.g. ad-hoc
2547
- # notes left by the operator) are out of contract scope.
2548
- continue
2549
- if match.group("task_type") != task_type:
2550
- # Cross-phase artifacts shouldn't appear here; skip rather
2551
- # than fail to keep the check focused on the current phase.
2552
- continue
2553
- if run_seq is not None and match.group("seq") != run_seq:
2554
- # A prior run's artifact. Its contract was judged when it ran.
2555
- continue
2556
-
2557
- worker_role = match.group("worker")
2558
- seq = match.group("seq")
2559
- rel = path.name
2560
- try:
2561
- content = path.read_text()
2562
- except OSError as exc:
2563
- failures.append(f"worker-results file unreadable: {rel} ({exc})")
2564
- continue
2400
+ """Enforce the worker audit sidecar contract at Phase 7.
2565
2401
 
2566
- if _READING_CONFIRMATION_HEADING_RE.search(content) is not None:
2567
- failures.append(
2568
- f"worker-results file `{rel}` contains a `## 0. Reading "
2569
- f"Confirmation` heading — that block moved to the audit "
2570
- f"sidecar (`{worker_role}-audit-{task_type}-{seq}.md`). "
2571
- f"Remove the §0 heading + body from the main file and "
2572
- f"write a fresh sidecar."
2573
- )
2574
-
2575
- audit_path = worker_results_dir / f"{worker_role}-audit-{task_type}-{seq}.md"
2576
- if not audit_path.exists():
2577
- failures.append(
2578
- f"worker `{worker_role}` produced `{rel}` but no audit sidecar "
2579
- f"at `{audit_path.name}` — the sidecar must carry the Reading "
2580
- f"Confirmation block (one short line per input file). Workers "
2581
- f"write this in the same step as the main worker-results file."
2582
- )
2583
- continue
2584
-
2585
- _validate_worker_evidence_read_ledger(
2586
- report_path=report_path,
2587
- worker_role=worker_role,
2588
- task_type=task_type,
2589
- seq=seq,
2590
- result_name=rel,
2591
- result_content=content,
2592
- audit_path=audit_path,
2593
- failures=failures,
2594
- )
2402
+ The rules themselves live in `okstra_ctl.worker_audit_ledger` so that
2403
+ `okstra worker-audit-check` can apply the identical checks mid-run, while
2404
+ the worker session is still alive and can fix its own citations. All this
2405
+ wrapper adds is the Phase 7 anchor: the run directory and this run's seq,
2406
+ both read off the report path.
2407
+ """
2408
+ failures.extend(check_worker_results_audit(
2409
+ # `report_path` is `runs/<task-type>/reports/final-report-...md`.
2410
+ report_path.parent.parent,
2411
+ task_type,
2412
+ _report_run_seq(report_path),
2413
+ ))
2595
2414
 
2596
2415
 
2597
2416
  def validate_team_state_usage(team_state: dict, failures: list[str]) -> None:
@@ -2972,6 +2791,68 @@ def _route_target_matches(value: Any, target: str, *, command: bool) -> bool:
2972
2791
  )
2973
2792
 
2974
2793
 
2794
+ def _upstream_by_candidate(candidates: list[Any]) -> dict[str, list[str]]:
2795
+ upstream: dict[str, list[str]] = {}
2796
+ for candidate in candidates:
2797
+ if not isinstance(candidate, Mapping):
2798
+ continue
2799
+ candidate_id = candidate.get("id")
2800
+ declared = candidate.get("downstreamOf")
2801
+ if isinstance(candidate_id, str) and isinstance(declared, list):
2802
+ upstream[candidate_id] = [row for row in declared if isinstance(row, str)]
2803
+ return upstream
2804
+
2805
+
2806
+ def _chain_cycle(upstream: dict[str, list[str]]) -> list[str]:
2807
+ """The first cycle reachable through `downstreamOf`, as the ids that form it.
2808
+
2809
+ A cycle is a diagnosis that says each step is caused by the next, so it
2810
+ names no first cause. It also hangs the figure's layering, which relaxes
2811
+ until depths settle.
2812
+ """
2813
+ settled: set[str] = set()
2814
+ for start in sorted(upstream):
2815
+ stack = [start]
2816
+ on_path: list[str] = []
2817
+ while stack:
2818
+ current = stack.pop()
2819
+ if current in on_path:
2820
+ return on_path[on_path.index(current):] + [current]
2821
+ if current in settled or current not in upstream:
2822
+ continue
2823
+ on_path.append(current)
2824
+ stack.extend(upstream[current])
2825
+ settled.update(on_path)
2826
+ return []
2827
+
2828
+
2829
+ def _validate_cause_chain(
2830
+ candidates: list[Any], candidate_ids: set[str], failures: list[str]
2831
+ ) -> None:
2832
+ """`downstreamOf` must name a sibling candidate, and never itself."""
2833
+ upstream = _upstream_by_candidate(candidates)
2834
+ for candidate_id in sorted(upstream):
2835
+ unknown = sorted(set(upstream[candidate_id]) - candidate_ids)
2836
+ if unknown:
2837
+ failures.append(
2838
+ f"final-report data.json: {candidate_id}.downstreamOf names "
2839
+ "unknown cause candidate(s): " + ", ".join(unknown) + "."
2840
+ )
2841
+ if candidate_id in upstream[candidate_id]:
2842
+ failures.append(
2843
+ f"final-report data.json: {candidate_id}.downstreamOf names itself."
2844
+ )
2845
+ cycle = _chain_cycle(
2846
+ {key: [row for row in value if row in candidate_ids] for key, value in upstream.items()}
2847
+ )
2848
+ if cycle:
2849
+ failures.append(
2850
+ "final-report data.json: cause candidates form a downstreamOf cycle: "
2851
+ + " -> ".join(cycle)
2852
+ + "."
2853
+ )
2854
+
2855
+
2975
2856
  def _validate_error_analysis_consistency(
2976
2857
  data: Mapping[str, Any], failures: list[str]
2977
2858
  ) -> None:
@@ -3018,6 +2899,8 @@ def _validate_error_analysis_consistency(
3018
2899
  + "."
3019
2900
  )
3020
2901
 
2902
+ _validate_cause_chain(candidates, set(candidate_ids), failures)
2903
+
3021
2904
  routing_value = error_analysis.get("routing")
3022
2905
  routing = routing_value if isinstance(routing_value, Mapping) else {}
3023
2906
  target = routing.get("nextTaskType")
@@ -3243,6 +3126,9 @@ def validate_final_report_data(
3243
3126
  _validate_verdict_card_fields(data, failures)
3244
3127
  # Phase-agnostic: the coverage critic runs in every finding-producing phase.
3245
3128
  _validate_unverified_critic_gaps_recorded(data, failures)
3129
+ # Called here rather than from a task-type branch: four profiles raise
3130
+ # clarification rows, and the gate scopes itself by task type internally.
3131
+ _validate_clarification_options(data, failures)
3246
3132
 
3247
3133
  task_type = (data.get("header") or {}).get("taskType")
3248
3134
  _validate_verifier_fail_blocks_verdict(data, failures)
@@ -3267,6 +3153,8 @@ def validate_final_report_data(
3267
3153
  print(f"validate-run: warning: {warning}", file=sys.stderr)
3268
3154
  _validate_supersession_ledger(data, failures)
3269
3155
  _validate_clarification_evidence_note(data, failures)
3156
+ _validate_approval_clarification_backtrace(data, failures)
3157
+ _validate_rerun_guidance(data, failures)
3270
3158
  _validate_variation_point_analysis(
3271
3159
  (data.get("implementationPlanning") or {}).get("variationPointAnalysis"),
3272
3160
  resolve_architecture(_project_root_from_report(report_path)),
@@ -3594,6 +3482,38 @@ def _self_fix_budget_exhausted(pbv: dict) -> bool:
3594
3482
  )
3595
3483
 
3596
3484
 
3485
+ def _state_classification(item: dict, gate_class: str) -> str:
3486
+ """This item's `planItems[].rounds[].classification` for the state file.
3487
+
3488
+ The gate classifier deliberately folds `partial-consensus` and
3489
+ `dissent-isolated` into `has-dissent` — only the majority-disagree boundary
3490
+ moves the gate. The state file records the finer label, and the information
3491
+ to recover it is in the same verdicts, so the mapping lives beside the
3492
+ classifier rather than being re-invented by each lead.
3493
+
3494
+ *gate_class* is passed in rather than recomputed so that the caller's
3495
+ effective classification — which may have been downgraded by
3496
+ `_is_dissent_downgraded` — is the one this translates.
3497
+
3498
+ `contested` never appears: it is only meaningful at `maxRounds > 1`, and at
3499
+ the default `maxRounds=1` the round protocol folds any otherwise-unresolved
3500
+ item into `partial-consensus`.
3501
+ """
3502
+ if gate_class == "all-non-result":
3503
+ # No non-error vote at all is the `needs-reverify` shape taken to its
3504
+ # limit — "fewer than 2 participating votes" covers zero.
3505
+ return "needs-reverify"
3506
+ if gate_class != "has-dissent":
3507
+ return gate_class
3508
+ dissenting = sum(
3509
+ 1
3510
+ for vote in (item.get("verdicts") or [])
3511
+ if isinstance(vote, dict)
3512
+ and str(vote.get("verdict") or "").strip().upper() == "DISAGREE"
3513
+ )
3514
+ return "dissent-isolated" if dissenting == 1 else "partial-consensus"
3515
+
3516
+
3597
3517
  def _is_dissent_downgraded(item: dict, pbv: dict) -> bool:
3598
3518
  """Whether a surviving `majority-disagree` item stops blocking approval.
3599
3519
 
@@ -4116,6 +4036,200 @@ def _validate_clarification_evidence_note(data: dict, failures: list[str]) -> No
4116
4036
  )
4117
4037
 
4118
4038
 
4039
+ _CLARIFICATION_OPTION_SCHEMA_VERSION = "2.0"
4040
+ # The four profiles that read `_clarification-recommendation.md`. Unlike the
4041
+ # evidence-note gate above — which is called from inside the
4042
+ # `implementation-planning` branch — this one is called phase-agnostically, so
4043
+ # the task-type filter here is the only thing scoping it.
4044
+ _CLARIFICATION_OPTION_TASK_TYPES = frozenset({
4045
+ "error-analysis",
4046
+ "implementation-planning",
4047
+ "improvement-discovery",
4048
+ "requirements-discovery",
4049
+ })
4050
+ # `in-repo` and `cross-repo` answer the same question, so exactly one may hold.
4051
+ _REACH_TOKENS = frozenset({"in-repo", "cross-repo"})
4052
+ _LEGACY_EXPECTED_FORM_RE = re.compile(r"\b(?:Recommended|Alternatives):")
4053
+
4054
+
4055
+ def _validate_clarification_options(data: dict, failures: list[str]) -> None:
4056
+ """A `decision` row must carry its choices as data, not as prose.
4057
+
4058
+ The choices used to live inside the `expectedForm` string, where two
4059
+ separate parsers split them differently and neither was checked — the board
4060
+ the user picked from could disagree with the board the report meant.
4061
+ Structured options remove the parsing; this gate keeps the structure
4062
+ honest. Whether an impact claim is *true* is the adversarial round's job,
4063
+ exactly as with `Evidence checked:`.
4064
+
4065
+ schema-v1 is exempt because it cannot comply: it keeps clarifications as a
4066
+ Markdown table of strings and its schema forbids an `options` property, so
4067
+ demanding one would fail every v1 `decision` row for a structure the format
4068
+ has no place to hold.
4069
+ """
4070
+ if data.get("schemaVersion") != _CLARIFICATION_OPTION_SCHEMA_VERSION:
4071
+ return
4072
+ task_type = (data.get("header") or {}).get("taskType")
4073
+ if task_type not in _CLARIFICATION_OPTION_TASK_TYPES:
4074
+ return
4075
+ for row in data.get("clarificationItems") or []:
4076
+ if not isinstance(row, dict) or row.get("kind") != "decision":
4077
+ continue
4078
+ row_id = str(row.get("id") or "<unknown>")
4079
+ _validate_option_set(row.get("options"), row_id, failures)
4080
+ if _LEGACY_EXPECTED_FORM_RE.search(str(row.get("expectedForm") or "")):
4081
+ failures.append(
4082
+ f"final-report data.json: clarification `{row_id}` still encodes "
4083
+ "its choices in `expectedForm` (`Recommended:` / `Alternatives:`). "
4084
+ "Choices belong in `options[]`; `expectedForm` states only the "
4085
+ "shape of the answer. Two sources for one fact leave consumers "
4086
+ "disagreeing about which is authoritative."
4087
+ )
4088
+
4089
+
4090
+ def _validate_option_set(
4091
+ options: object, row_id: str, failures: list[str]
4092
+ ) -> None:
4093
+ """Check one `decision` row's `options[]` for pickability and reach."""
4094
+ if not isinstance(options, list) or len(options) < 2:
4095
+ failures.append(
4096
+ f"final-report data.json: clarification `{row_id}` is a `decision` "
4097
+ "but does not offer at least two `options[]`. A decision the user "
4098
+ "cannot choose between is not a decision."
4099
+ )
4100
+ return
4101
+ entries = [option for option in options if isinstance(option, dict)]
4102
+ recommended = sum(1 for option in entries if option.get("role") == "recommended")
4103
+ if recommended != 1:
4104
+ failures.append(
4105
+ f"final-report data.json: clarification `{row_id}` must carry exactly "
4106
+ f"one `role: recommended` option (found {recommended}). The reader "
4107
+ "needs to know which answer the run stands behind."
4108
+ )
4109
+ for index, option in enumerate(entries):
4110
+ tokens = option.get("scopeImpact")
4111
+ reach = (
4112
+ [token for token in tokens if token in _REACH_TOKENS]
4113
+ if isinstance(tokens, list)
4114
+ else []
4115
+ )
4116
+ if len(reach) != 1:
4117
+ failures.append(
4118
+ f"final-report data.json: clarification `{row_id}` option "
4119
+ f"[{index}] must declare exactly one reach token — `in-repo` or "
4120
+ f"`cross-repo` (found {len(reach)}). The reader cannot weigh an "
4121
+ "option whose reach is unstated or self-contradictory."
4122
+ )
4123
+
4124
+
4125
+ def _has_clarification_backtrace(
4126
+ row_id: str, plan_items: object, coverage: object
4127
+ ) -> bool:
4128
+ """Whether the plan records anything this clarification blocks.
4129
+
4130
+ Two link shapes, both authored by the same run: the `P-*` plan item that
4131
+ carries the `clarificationId`, and the requirement-coverage row blocked on
4132
+ the id. `incremental-scope` resolves impacted stages from exactly these
4133
+ two, and the coverage side goes through its predicate so the gate and the
4134
+ resolver cannot disagree about what counts as a link.
4135
+ """
4136
+ if isinstance(plan_items, list) and any(
4137
+ isinstance(item, dict) and item.get("clarificationId") == row_id
4138
+ for item in plan_items
4139
+ ):
4140
+ return True
4141
+ return isinstance(coverage, list) and any(
4142
+ coverage_row_blocked_on(row, row_id) for row in coverage
4143
+ )
4144
+
4145
+
4146
+ def _validate_approval_clarification_backtrace(
4147
+ data: dict, failures: list[str]
4148
+ ) -> None:
4149
+ """An approval blocker must record what it blocks.
4150
+
4151
+ `_validate_plan_body_clarification_matching` already walks the other
4152
+ direction — a majority-disagree plan item must cite a `blocks: approval`
4153
+ row. Nothing walked this way, so a row could withhold approval while
4154
+ recording no blast radius at all. The cost lands on the re-run:
4155
+ `incremental-scope` resolves impacted stages from these links and treats an
4156
+ id that traces to no stage as grounds to re-verify everything, so one
4157
+ unlinked blocker turns a narrow re-run into a full one.
4158
+ """
4159
+ if (data.get("header") or {}).get("taskType") != "implementation-planning":
4160
+ return
4161
+ planning = data.get("implementationPlanning")
4162
+ if not isinstance(planning, dict):
4163
+ return
4164
+ coverage = planning.get("requirementCoverage")
4165
+ verification = planning.get("planBodyVerification")
4166
+ plan_items = (
4167
+ verification.get("planItems") if isinstance(verification, dict) else None
4168
+ )
4169
+ for row in data.get("clarificationItems") or []:
4170
+ if not isinstance(row, dict) or row.get("blocks") != "approval":
4171
+ continue
4172
+ row_id = str(row.get("id") or "<unknown>")
4173
+ if _has_clarification_backtrace(row_id, plan_items, coverage):
4174
+ continue
4175
+ failures.append(
4176
+ f"final-report data.json: clarification `{row_id}` blocks approval "
4177
+ "but has no back-trace into the plan — no plan item carries it as "
4178
+ "`clarificationId`, and no requirement-coverage row is `blocked "
4179
+ f"{row_id}` in its `status` or `approvalDisposition`. An item that "
4180
+ "withholds approval without recording what it affects forces the "
4181
+ "next re-run to re-verify everything."
4182
+ )
4183
+
4184
+
4185
+ _RERUN_FLAG = "--answered-clarifications"
4186
+
4187
+
4188
+ def _next_step_texts(steps: object) -> list[str]:
4189
+ """Every reader-visible string in `recommendedNextSteps`, prose and command."""
4190
+ texts: list[str] = []
4191
+ for step in steps if isinstance(steps, list) else []:
4192
+ if not isinstance(step, dict):
4193
+ continue
4194
+ texts.append(str(step.get("text") or ""))
4195
+ for command in step.get("commands") or []:
4196
+ if isinstance(command, dict):
4197
+ texts.append(str(command.get("claudeCode") or ""))
4198
+ texts.append(str(command.get("terminal") or ""))
4199
+ return texts
4200
+
4201
+
4202
+ def _validate_rerun_guidance(data: dict, failures: list[str]) -> None:
4203
+ """A report that withholds approval must say how to come back from it.
4204
+
4205
+ The way forward — answer the blockers, then re-run carrying those ids —
4206
+ lived only in the lead prompt, which is read *after* the next run has
4207
+ already started. The person who has to act reads the report instead, and it
4208
+ told them nothing about the next command. Requiring the flag by name is a
4209
+ low bar deliberately: it does not check that the rest of the step is right,
4210
+ only that the report stops leaving the reader to work the mechanics out.
4211
+ """
4212
+ if (data.get("header") or {}).get("taskType") != "implementation-planning":
4213
+ return
4214
+ has_blocker = any(
4215
+ isinstance(row, dict) and row.get("blocks") == "approval"
4216
+ for row in data.get("clarificationItems") or []
4217
+ )
4218
+ if not has_blocker:
4219
+ return
4220
+ if any(_RERUN_FLAG in text for text in _next_step_texts(
4221
+ data.get("recommendedNextSteps")
4222
+ )):
4223
+ return
4224
+ failures.append(
4225
+ "final-report data.json: this plan withholds approval on a "
4226
+ "`blocks: approval` clarification, but no `recommendedNextSteps` entry "
4227
+ f"tells the reader how to resume — name the `{_RERUN_FLAG}` re-run in a "
4228
+ "step's `text` or one of its `commands`. `okstra recap assemble` "
4229
+ "prints the exact ids and flag value once the answers are recorded."
4230
+ )
4231
+
4232
+
4119
4233
  def _validate_self_fix_grouping(data: dict, failures: list[str]) -> None:
4120
4234
  """A self-fix round must be instructed by cause, not as a flat item list.
4121
4235
 
@@ -5866,6 +5980,21 @@ def validate_plan_body_section(
5866
5980
  return [*_detect_self_fix_recurrence(pbv), *_detect_uniform_verifier(pbv)]
5867
5981
 
5868
5982
 
5983
+ def _gate_summary_item(item: dict, pbv: dict) -> dict:
5984
+ """One `gate.items[]` row: the gate class plus its state-file counterpart."""
5985
+ classification = (
5986
+ "has-dissent"
5987
+ if _is_dissent_downgraded(item, pbv)
5988
+ else _classify_plan_item_gate(item)
5989
+ )
5990
+ return {
5991
+ "id": item.get("id"),
5992
+ "classification": classification,
5993
+ "stateClassification": _state_classification(item, classification),
5994
+ "correctnessCritical": _is_correctness_critical(item),
5995
+ }
5996
+
5997
+
5869
5998
  def plan_body_gate_summary(data: dict) -> dict | None:
5870
5999
  """The §5.5.9 gate as the round protocol's step 5 needs it — per-item
5871
6000
  classification, the whole-gate value, and the `gateBlockedBy` causes, all
@@ -5887,13 +6016,7 @@ def plan_body_gate_summary(data: dict) -> dict | None:
5887
6016
  if recomputed is None:
5888
6017
  return None
5889
6018
  items = [
5890
- {
5891
- "id": item.get("id"),
5892
- "classification": "has-dissent"
5893
- if _is_dissent_downgraded(item, pbv)
5894
- else _classify_plan_item_gate(item),
5895
- "correctnessCritical": _is_correctness_critical(item),
5896
- }
6019
+ _gate_summary_item(item, pbv)
5897
6020
  for item in (pbv.get("planItems") or [])
5898
6021
  if isinstance(item, dict)
5899
6022
  ]