okstra 0.141.3 → 0.143.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/docs/architecture.md +11 -2
  2. package/docs/cli.md +15 -0
  3. package/docs/for-ai/skills/okstra-setup.md +8 -0
  4. package/docs/project-structure-overview.md +6 -0
  5. package/docs/task-process/error-analysis.md +9 -4
  6. package/package.json +1 -1
  7. package/runtime/BUILD.json +2 -2
  8. package/runtime/agents/workers/report-writer-worker.md +4 -2
  9. package/runtime/prompts/coding-preflight/architectures/hexagonal.md +3 -3
  10. package/runtime/prompts/coding-preflight/overview.md +1 -1
  11. package/runtime/prompts/lead/adapters/claude-code.md +2 -2
  12. package/runtime/prompts/lead/context-loader.md +2 -2
  13. package/runtime/prompts/lead/convergence.md +5 -2
  14. package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
  15. package/runtime/prompts/lead/plan-body-verification.md +20 -9
  16. package/runtime/prompts/lead/report-writer.md +4 -3
  17. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -0
  18. package/runtime/prompts/profiles/_common-contract.md +3 -1
  19. package/runtime/prompts/profiles/_implementation-deliverable.md +3 -3
  20. package/runtime/prompts/profiles/_implementation-diff-review.md +1 -1
  21. package/runtime/prompts/profiles/_implementation-verifier.md +2 -1
  22. package/runtime/prompts/profiles/error-analysis.md +5 -1
  23. package/runtime/prompts/profiles/forbidden-actions.json +0 -1
  24. package/runtime/prompts/profiles/implementation-planning.md +7 -2
  25. package/runtime/prompts/profiles/requirements-discovery.md +7 -0
  26. package/runtime/python/okstra_ctl/analysis_packet.py +28 -3
  27. package/runtime/python/okstra_ctl/brief_frontmatter.py +56 -0
  28. package/runtime/python/okstra_ctl/clarification_items.py +99 -5
  29. package/runtime/python/okstra_ctl/convergence_engine.py +66 -15
  30. package/runtime/python/okstra_ctl/paths.py +34 -7
  31. package/runtime/python/okstra_ctl/phase_cleanup.py +235 -0
  32. package/runtime/python/okstra_ctl/plan_items.py +38 -0
  33. package/runtime/python/okstra_ctl/run.py +81 -33
  34. package/runtime/python/okstra_ctl/schema_excerpt.py +5 -3
  35. package/runtime/python/okstra_ctl/wizard.py +18 -44
  36. package/runtime/python/okstra_ctl/worker_heartbeat.py +15 -5
  37. package/runtime/python/okstra_ctl/workflow.py +1 -1
  38. package/runtime/python/okstra_project/resolver.py +25 -0
  39. package/runtime/schemas/final-report-v1.0.schema.json +162 -4
  40. package/runtime/skills/okstra-run/SKILL.md +3 -1
  41. package/runtime/skills/okstra-setup/SKILL.md +3 -0
  42. package/runtime/skills/okstra-setup/references/project-config.md +47 -0
  43. package/runtime/templates/reports/final-report.template.md +51 -0
  44. package/runtime/templates/reports/i18n/en.json +32 -3
  45. package/runtime/templates/reports/i18n/ko.json +32 -3
  46. package/runtime/templates/reports/implementation-input.template.md +1 -2
  47. package/runtime/templates/reports/task-brief.template.md +1 -1
  48. package/runtime/validators/validate-brief.py +5 -1
  49. package/runtime/validators/validate-run.py +430 -48
  50. package/src/cli-registry.mjs +10 -0
  51. package/src/commands/execute/phase-cleanup.mjs +38 -0
@@ -8,9 +8,12 @@ import importlib.util
8
8
  import json
9
9
  import os
10
10
  import re
11
+ import shlex
11
12
  import sys
13
+ from collections.abc import Mapping
12
14
  from datetime import datetime, timezone
13
15
  from pathlib import Path
16
+ from typing import Any
14
17
 
15
18
  # Make the okstra packages importable however this validator is reached:
16
19
  # ``scripts/`` next to the repo checkout, ``python/`` next to the installed
@@ -34,6 +37,7 @@ except ImportError: # pragma: no cover — runtime guarantees this import
34
37
 
35
38
  from okstra_project import project_json_path # noqa: E402
36
39
  from okstra_project.dirs import tasks_root as _okstra_tasks_root # noqa: E402
40
+ from okstra_project.resolver import resolve_architecture # noqa: E402
37
41
 
38
42
  from okstra_ctl.conformance import ( # noqa: E402
39
43
  CAPABILITY_WHITELIST,
@@ -187,10 +191,29 @@ def advance_next_phase(
187
191
  return default_next_phase(current_phase)
188
192
 
189
193
 
194
+ def _error_analysis_next_phase(data: Mapping[str, Any]) -> str | None:
195
+ if not isinstance(data, Mapping):
196
+ return None
197
+ header = data.get("header")
198
+ if not isinstance(header, Mapping) or header.get("taskType") != "error-analysis":
199
+ return None
200
+ error_analysis = data.get("errorAnalysis")
201
+ if not isinstance(error_analysis, Mapping):
202
+ return None
203
+ routing = error_analysis.get("routing")
204
+ if not isinstance(routing, Mapping):
205
+ return None
206
+ target = routing.get("nextTaskType")
207
+ if target in {"error-analysis", "implementation-planning"}:
208
+ return str(target)
209
+ return None
210
+
211
+
190
212
  def update_workflow_metadata(
191
213
  run_manifest: dict,
192
214
  task_manifest: dict,
193
215
  validation_status: str,
216
+ report_data: Mapping[str, Any] | None = None,
194
217
  ) -> None:
195
218
  workflow = task_manifest.get("workflow", {})
196
219
  if not isinstance(workflow, dict):
@@ -220,7 +243,14 @@ def update_workflow_metadata(
220
243
  # Validation just passed → actively advance to the next phase in
221
244
  # the sequence rather than preserving a stale value that may equal
222
245
  # current_phase (which would cause the lifecycle pointer to stall).
223
- next_recommended_phase = advance_next_phase(current_phase, phase_sequence)
246
+ report_next_phase = (
247
+ _error_analysis_next_phase(report_data or {})
248
+ if current_phase == "error-analysis"
249
+ else None
250
+ )
251
+ next_recommended_phase = report_next_phase or advance_next_phase(
252
+ current_phase, phase_sequence
253
+ )
224
254
  else:
225
255
  current_phase_state = "blocked"
226
256
  if current_phase:
@@ -321,6 +351,7 @@ def update_validation_metadata(
321
351
  task_manifest: dict,
322
352
  validation_status: str,
323
353
  failures: list[str],
354
+ report_data: Mapping[str, Any] | None = None,
324
355
  ) -> None:
325
356
  checked_at = utc_now()
326
357
 
@@ -349,7 +380,12 @@ def update_validation_metadata(
349
380
  task_manifest["currentStatus"] = (
350
381
  "completed" if validation_status == "passed" else "contract-violated"
351
382
  )
352
- update_workflow_metadata(run_manifest, task_manifest, validation_status)
383
+ update_workflow_metadata(
384
+ run_manifest,
385
+ task_manifest,
386
+ validation_status,
387
+ report_data=report_data,
388
+ )
353
389
 
354
390
 
355
391
  def extract_contract(
@@ -1644,7 +1680,10 @@ def _validate_conformance(
1644
1680
 
1645
1681
 
1646
1682
  def _check_incremental_audit_block(
1647
- report_path: Path, content: str, failures: list[str]
1683
+ report_path: Path,
1684
+ content: str,
1685
+ failures: list[str],
1686
+ report_data: Mapping[str, Any] | None = None,
1648
1687
  ) -> None:
1649
1688
  """Enforce the Section 0 incremental audit block.
1650
1689
 
@@ -1656,7 +1695,11 @@ def _check_incremental_audit_block(
1656
1695
  decision are exempt, so this never conflicts with the empty-Section-0
1657
1696
  stub rule (that fires only when NO carry-in was provided at all).
1658
1697
  """
1659
- data = _load_final_report_data(report_path)
1698
+ data = (
1699
+ report_data
1700
+ if report_data is not None
1701
+ else _load_final_report_data(report_path)
1702
+ )
1660
1703
  planning = data.get("implementationPlanning")
1661
1704
  if not isinstance(planning, dict):
1662
1705
  return
@@ -1688,6 +1731,7 @@ def validate_report(
1688
1731
  *,
1689
1732
  allow_unavailable_token_usage: bool = False,
1690
1733
  unavailable_ok_labels: frozenset[str] = frozenset(),
1734
+ report_data: Mapping[str, Any] | None = None,
1691
1735
  ) -> None:
1692
1736
  if not report_path.exists():
1693
1737
  failures.append(f"final report is missing: {report_path}")
@@ -1778,7 +1822,12 @@ def validate_report(
1778
1822
 
1779
1823
  # Incremental audit block — an `incremental`-mode re-run must expose its
1780
1824
  # scope decision in Section 0 so the narrowed re-verification is auditable.
1781
- _check_incremental_audit_block(report_path, content, failures)
1825
+ _check_incremental_audit_block(
1826
+ report_path,
1827
+ content,
1828
+ failures,
1829
+ report_data=report_data,
1830
+ )
1782
1831
 
1783
1832
  # Deprecated section headings — pre-1.0 hard removal.
1784
1833
  for pattern, remedy in _DEPRECATED_FINAL_REPORT_PATTERNS:
@@ -2239,6 +2288,213 @@ def _validate_verdict_card_fields(data: dict, failures: list[str]) -> None:
2239
2288
  )
2240
2289
 
2241
2290
 
2291
+ def _route_target_matches(value: Any, target: str, *, command: bool) -> bool:
2292
+ if not isinstance(value, str):
2293
+ return False
2294
+ if not command:
2295
+ pattern = rf"(?<![\w-]){re.escape(target)}(?![\w-])"
2296
+ return re.search(pattern, value) is not None
2297
+ try:
2298
+ tokens = shlex.split(value)
2299
+ except ValueError:
2300
+ return False
2301
+ task_type_values: list[str] = []
2302
+ for index, token in enumerate(tokens):
2303
+ if token in {"task-type", "--task-type"}:
2304
+ task_type_values.append(
2305
+ tokens[index + 1] if index + 1 < len(tokens) else ""
2306
+ )
2307
+ for prefix in ("task-type=", "--task-type="):
2308
+ if token.startswith(prefix):
2309
+ task_type_values.append(token[len(prefix):])
2310
+ return bool(task_type_values) and all(
2311
+ task_type_value == target for task_type_value in task_type_values
2312
+ )
2313
+
2314
+
2315
+ def _validate_error_analysis_consistency(
2316
+ data: Mapping[str, Any], failures: list[str]
2317
+ ) -> None:
2318
+ error_analysis_value = data.get("errorAnalysis")
2319
+ error_analysis = (
2320
+ error_analysis_value if isinstance(error_analysis_value, Mapping) else {}
2321
+ )
2322
+ reproduction_value = error_analysis.get("reproduction")
2323
+ reproduction = (
2324
+ reproduction_value if isinstance(reproduction_value, Mapping) else {}
2325
+ )
2326
+ reproduction_status = reproduction.get("status")
2327
+ blocked_reason = reproduction.get("blockedReason")
2328
+ if reproduction_status == "blocked-before-repro":
2329
+ if not isinstance(blocked_reason, str) or not blocked_reason.strip():
2330
+ failures.append(
2331
+ "final-report data.json: blocked-before-repro requires a non-empty "
2332
+ "errorAnalysis.reproduction.blockedReason."
2333
+ )
2334
+ elif blocked_reason != "":
2335
+ failures.append(
2336
+ "final-report data.json: errorAnalysis.reproduction.blockedReason "
2337
+ "must be exactly empty unless status is blocked-before-repro."
2338
+ )
2339
+
2340
+ candidates_value = error_analysis.get("causeCandidates")
2341
+ candidates = candidates_value if isinstance(candidates_value, list) else []
2342
+ candidate_ids: list[str] = []
2343
+ for candidate in candidates:
2344
+ if not isinstance(candidate, Mapping):
2345
+ continue
2346
+ candidate_id = candidate.get("id")
2347
+ if isinstance(candidate_id, str):
2348
+ candidate_ids.append(candidate_id)
2349
+ duplicate_ids = sorted(
2350
+ candidate_id
2351
+ for candidate_id in set(candidate_ids)
2352
+ if candidate_ids.count(candidate_id) > 1
2353
+ )
2354
+ if duplicate_ids:
2355
+ failures.append(
2356
+ "final-report data.json: duplicate cause candidate id(s): "
2357
+ + ", ".join(duplicate_ids)
2358
+ + "."
2359
+ )
2360
+
2361
+ routing_value = error_analysis.get("routing")
2362
+ routing = routing_value if isinstance(routing_value, Mapping) else {}
2363
+ target = routing.get("nextTaskType")
2364
+ leading_cause_id = routing.get("leadingCauseId")
2365
+ candidate_id_set = set(candidate_ids)
2366
+ if target == "implementation-planning":
2367
+ if not candidates:
2368
+ failures.append(
2369
+ "final-report data.json: implementation-planning routing requires "
2370
+ "at least one cause candidate."
2371
+ )
2372
+ if (
2373
+ not isinstance(leading_cause_id, str)
2374
+ or leading_cause_id not in candidate_id_set
2375
+ ):
2376
+ failures.append(
2377
+ "final-report data.json: implementation-planning routing "
2378
+ "leadingCauseId must reference a cause candidate."
2379
+ )
2380
+ elif target == "error-analysis" and (
2381
+ not isinstance(leading_cause_id, str)
2382
+ or (leading_cause_id != "" and leading_cause_id not in candidate_id_set)
2383
+ ):
2384
+ failures.append(
2385
+ "final-report data.json: error-analysis routing leadingCauseId must be "
2386
+ "empty or reference a cause candidate."
2387
+ )
2388
+
2389
+ expected_direction = {
2390
+ "implementation-planning": "begin-planning",
2391
+ "error-analysis": "continue-investigation",
2392
+ }.get(target)
2393
+ verdict_card_value = data.get("verdictCard")
2394
+ verdict_card = (
2395
+ verdict_card_value if isinstance(verdict_card_value, Mapping) else {}
2396
+ )
2397
+ final_verdict_value = data.get("finalVerdict")
2398
+ final_verdict = (
2399
+ final_verdict_value if isinstance(final_verdict_value, Mapping) else {}
2400
+ )
2401
+ if expected_direction:
2402
+ for field_name, verdict in (
2403
+ ("verdictCard", verdict_card),
2404
+ ("finalVerdict", final_verdict),
2405
+ ):
2406
+ if verdict.get("direction") != expected_direction:
2407
+ failures.append(
2408
+ f"final-report data.json: {field_name}.direction must be "
2409
+ f"`{expected_direction}` for `{target}` routing."
2410
+ )
2411
+
2412
+ follow_up_tasks_value = data.get("followUpTasks")
2413
+ follow_up_tasks = (
2414
+ follow_up_tasks_value if isinstance(follow_up_tasks_value, list) else []
2415
+ )
2416
+ continuations = [
2417
+ row
2418
+ for row in follow_up_tasks
2419
+ if isinstance(row, Mapping) and row.get("origin") == "phase-continuation"
2420
+ ]
2421
+ if len(continuations) != 1:
2422
+ failures.append(
2423
+ "final-report data.json: followUpTasks must contain exactly one "
2424
+ "phase-continuation row."
2425
+ )
2426
+ else:
2427
+ continuation = continuations[0]
2428
+ frontmatter_value = data.get("frontmatter")
2429
+ frontmatter = (
2430
+ frontmatter_value if isinstance(frontmatter_value, Mapping) else {}
2431
+ )
2432
+ task_id = frontmatter.get("taskId")
2433
+ if continuation.get("suggestedTaskType") != target:
2434
+ failures.append(
2435
+ "final-report data.json: phase-continuation suggestedTaskType "
2436
+ "must match the routing target."
2437
+ )
2438
+ if continuation.get("newTaskId") != task_id:
2439
+ failures.append(
2440
+ "final-report data.json: phase-continuation newTaskId must match "
2441
+ "frontmatter.taskId."
2442
+ )
2443
+ if continuation.get("priority") != "P0":
2444
+ failures.append(
2445
+ "final-report data.json: phase-continuation priority must be P0."
2446
+ )
2447
+ if continuation.get("autoSpawn") != "no":
2448
+ failures.append(
2449
+ "final-report data.json: phase-continuation autoSpawn must be no."
2450
+ )
2451
+
2452
+ if isinstance(target, str) and target in {
2453
+ "error-analysis",
2454
+ "implementation-planning",
2455
+ }:
2456
+ for field_name, value in (
2457
+ ("verdictCard.nextStep", verdict_card.get("nextStep")),
2458
+ ("finalVerdict.nextStep", final_verdict.get("nextStep")),
2459
+ ):
2460
+ if not _route_target_matches(value, target, command=False):
2461
+ failures.append(
2462
+ f"final-report data.json: {field_name} must contain routing "
2463
+ f"target `{target}`."
2464
+ )
2465
+
2466
+ next_steps_value = data.get("recommendedNextSteps")
2467
+ next_steps = next_steps_value if isinstance(next_steps_value, list) else []
2468
+ first_step = (
2469
+ next_steps[0]
2470
+ if next_steps and isinstance(next_steps[0], Mapping)
2471
+ else {}
2472
+ )
2473
+ step_text = first_step.get("text")
2474
+ if not _route_target_matches(step_text, target, command=False):
2475
+ failures.append(
2476
+ "final-report data.json: recommendedNextSteps[0].text must contain "
2477
+ f"routing target `{target}`."
2478
+ )
2479
+ commands_value = first_step.get("commands")
2480
+ commands = commands_value if isinstance(commands_value, list) else []
2481
+ if not commands:
2482
+ failures.append(
2483
+ "final-report data.json: recommendedNextSteps[0].commands must "
2484
+ "contain at least one command."
2485
+ )
2486
+ for index, command_value in enumerate(commands):
2487
+ command = command_value if isinstance(command_value, Mapping) else {}
2488
+ for command_field in ("claudeCode", "terminal"):
2489
+ value = command.get(command_field)
2490
+ if not _route_target_matches(value, target, command=True):
2491
+ failures.append(
2492
+ "final-report data.json: recommendedNextSteps[0].commands"
2493
+ f"[{index}].{command_field} must contain routing target "
2494
+ f"`{target}`."
2495
+ )
2496
+
2497
+
2242
2498
  def _load_final_report_data(report_path: Path) -> dict:
2243
2499
  """Best-effort parse of the final-report data.json sibling. Returns {} when
2244
2500
  absent or unparseable — those conditions are already surfaced as failures by
@@ -2257,7 +2513,7 @@ def validate_final_report_data(
2257
2513
  failures: list[str],
2258
2514
  *,
2259
2515
  report_contracts: set[str] | None = None,
2260
- ) -> None:
2516
+ ) -> Mapping[str, Any] | None:
2261
2517
  """Validate the final-report data.json against the v1.0 schema.
2262
2518
 
2263
2519
  The data.json is the source-of-truth that the renderer reads to
@@ -2270,6 +2526,8 @@ def validate_final_report_data(
2270
2526
  Missing data.json is reported as a single failure rather than a
2271
2527
  cascade of substring failures — that points the writer at the right
2272
2528
  fix (write the data.json) instead of futilely editing the markdown.
2529
+ The returned mapping is the exact loaded snapshot consumed by later
2530
+ finalization checks and workflow persistence.
2273
2531
  """
2274
2532
  if schema_validate is None or load_schema is None:
2275
2533
  # Module-load fallback path; should never fire in a real install.
@@ -2314,7 +2572,9 @@ def validate_final_report_data(
2314
2572
  _validate_verifier_fail_blocks_verdict(data, failures)
2315
2573
  if task_type == "implementation":
2316
2574
  _validate_stage_carry_sidecar_exists(data, report_path, failures)
2317
- if task_type == "final-verification":
2575
+ if task_type == "error-analysis":
2576
+ _validate_error_analysis_consistency(data, failures)
2577
+ elif task_type == "final-verification":
2318
2578
  _validate_final_verification_consistency(data, failures)
2319
2579
  _validate_verified_row_recorded(data, report_path, failures)
2320
2580
  elif task_type == "implementation-planning":
@@ -2343,6 +2603,11 @@ def validate_final_report_data(
2343
2603
  _validate_round_recorded_verdicts(data, failures)
2344
2604
  _validate_verdicts_match_current_subjects(data, failures)
2345
2605
  _validate_plan_item_extraction_completeness(data, failures)
2606
+ _validate_variation_point_analysis(
2607
+ (data.get("implementationPlanning") or {}).get("variationPointAnalysis"),
2608
+ resolve_architecture(_project_root_from_report(report_path)),
2609
+ failures,
2610
+ )
2346
2611
  _validate_plan_item_subject_substance(data, failures)
2347
2612
  _validate_plan_body_clarification_matching(data, failures)
2348
2613
  _validate_disagree_has_fixability(data, failures)
@@ -2358,6 +2623,8 @@ def validate_final_report_data(
2358
2623
  for warning in warnings:
2359
2624
  print(f"validate-run: warning: {warning}", file=sys.stderr)
2360
2625
 
2626
+ return data
2627
+
2361
2628
 
2362
2629
  # path:line (foo.service.ts:268), an in-report ID (C-001 / E-006 / P-004), a
2363
2630
  # namespaced audit ref (claude:F-005), or a §section reference (§5.4) — any one
@@ -2524,10 +2791,16 @@ _PLAN_GATE_RANK = {
2524
2791
 
2525
2792
  # Breakage kinds where a single DISAGREE blocks the gate on its own (no majority
2526
2793
  # needed), because the defect is concrete, safety-critical, and adversarially
2527
- # verifiable: `a` = cited path/symbol mismatch, `d` = rollback violates
2528
- # commit/dependency order. `b`/`c`/`e` still need a majority — `b` in particular
2529
- # is prone to planning-vs-implementation environment false positives.
2530
- _SINGLE_VOTE_BLOCKING_KINDS = {"a", "d"}
2794
+ # verifiable: `a` = cited path/symbol mismatch. `b`/`c`/`e` still need a
2795
+ # majority — `b` in particular is prone to planning-vs-implementation
2796
+ # environment false positives.
2797
+ _SINGLE_VOTE_BLOCKING_KINDS = {"a"}
2798
+
2799
+ # Rollback ordering (`d`) is executed by a human, not by okstra's workers or
2800
+ # verifiers, so a rollback-ordering dissent is recorded but never gates
2801
+ # approval: it is dropped from the blocking-disagree tally entirely, so an item
2802
+ # whose only DISAGREEs are advisory-only can never rise above `has-dissent`.
2803
+ _ADVISORY_ONLY_KINDS = {"d"}
2531
2804
 
2532
2805
  # Stop reasons that justify promoting a still-broken planner-fixable item to the
2533
2806
  # user: the self-fix budget ran out, or a round produced no net resolution so
@@ -2535,6 +2808,20 @@ _SINGLE_VOTE_BLOCKING_KINDS = {"a", "d"}
2535
2808
  _SELF_FIX_EXHAUSTED_REASONS = frozenset({"max-rounds-reached", "no-progress"})
2536
2809
 
2537
2810
 
2811
+ def _is_variation_point_item(item: dict) -> bool:
2812
+ """Whether this is a `P-Var-*` variation-point item, which is majority-gated
2813
+ (`prompts/lead/plan-body-verification.md` "`P-Var-<N>` … is majority-gated").
2814
+ Whether a behavior has two implementations, and whether the plan extracted the
2815
+ right interface for it, is a design judgement — it lacks the concrete certainty
2816
+ of kind `a`, where a verifier points at two spelled-out references that
2817
+ contradict each other. So kind `a` carries no extra weight on a P-Var item: it
2818
+ neither single-vote-blocks nor counts as correctness-critical, exactly like the
2819
+ `b` / `c` / `e` kinds the prompt routes P-Var defects to. Only a
2820
+ `majority-disagree` gates it — that part is unchanged.
2821
+ """
2822
+ return str(item.get("id") or "").upper().startswith("P-VAR")
2823
+
2824
+
2538
2825
  def _classify_plan_item_gate(item: dict) -> str:
2539
2826
  """Recompute one plan item's gate class from its per-worker verdicts,
2540
2827
  per `prompts/lead/plan-body-verification.md` "Round protocol". Returns one of
@@ -2556,15 +2843,32 @@ def _classify_plan_item_gate(item: dict) -> str:
2556
2843
  return "all-non-result"
2557
2844
  disagree = [(vd, bk) for (vd, bk) in non_error if vd == "DISAGREE"]
2558
2845
  agree = [(vd, bk) for (vd, bk) in non_error if vd in ("AGREE", "SUPPLEMENT")]
2559
- disagree_kinds = {bk for (_vd, bk) in disagree if bk}
2560
2846
  if not disagree:
2561
2847
  return "full-consensus"
2848
+ # Rollback is a human-run operation, so rollback dissent never blocks the
2849
+ # gate — closed from two angles so a verifier cannot re-block it by relabelling:
2850
+ # (1) a whole rollback plan item (`P-Rb-*`) is advisory regardless of
2851
+ # breakage kind — otherwise a `DISAGREE(b)` "rollback command is
2852
+ # ambiguous" would sail past the kind-`d` exemption and block;
2853
+ # (2) a rollback-ordering dissent (`d`) is advisory on ANY item, since a
2854
+ # rollback-order defect raised against a non-rollback item is still a
2855
+ # human-run concern.
2856
+ # Both are recorded as dissent and fold into `has-dissent`, never blocking.
2857
+ if str(item.get("id") or "").upper().startswith("P-RB"):
2858
+ return "has-dissent"
2859
+ blocking_disagree = [(vd, bk) for (vd, bk) in disagree if bk not in _ADVISORY_ONLY_KINDS]
2860
+ if not blocking_disagree:
2861
+ return "has-dissent"
2862
+ blocking_kinds = {bk for (_vd, bk) in blocking_disagree if bk}
2562
2863
  # Single-vote-blocking kinds: one confirmed DISAGREE on a concrete,
2563
2864
  # safety-critical, adversarially-verifiable defect is enough to block, even
2564
2865
  # in a two-worker roster — a lone correct dissent must not be outvoted here.
2565
- # `a`/`d` for any item; `f` only for P-Req items (requirement coverage).
2866
+ # `a` for any item except `P-Var-*` (majority-gated, see
2867
+ # `_is_variation_point_item`); `f` only for P-Req items (requirement coverage).
2566
2868
  is_req = str(item.get("id") or "").upper().startswith("P-REQ")
2567
- if disagree_kinds & _SINGLE_VOTE_BLOCKING_KINDS or (is_req and "f" in disagree_kinds):
2869
+ if not _is_variation_point_item(item) and (
2870
+ blocking_kinds & _SINGLE_VOTE_BLOCKING_KINDS or (is_req and "f" in blocking_kinds)
2871
+ ):
2568
2872
  # "One confirmed DISAGREE" presupposes the item was actually
2569
2873
  # cross-verified. When the peer returned a non-result nothing confirmed
2570
2874
  # the dissent, so blocking here would reproduce the same
@@ -2577,7 +2881,7 @@ def _classify_plan_item_gate(item: dict) -> str:
2577
2881
  # two participating votes, so a lone surviving DISAGREE (its peer returned a
2578
2882
  # non-result) does NOT block. That fixes the paradox where a worker failure
2579
2883
  # made the gate stricter than a healthy roster would.
2580
- if len(non_error) >= 2 and len(disagree) > len(agree):
2884
+ if len(non_error) >= 2 and len(blocking_disagree) > len(agree):
2581
2885
  return "majority-disagree"
2582
2886
  return "has-dissent"
2583
2887
 
@@ -2604,11 +2908,18 @@ def _has_planner_fixable_majority(item: dict) -> bool:
2604
2908
 
2605
2909
  def _is_correctness_critical(item: dict) -> bool:
2606
2910
  """Whether this item's defect would make `implementation` produce wrong or
2607
- unsafe code — the single-vote-blocking kinds `a` (cited path/symbol
2608
- mismatch) and `d` (rollback violates commit order) on any item, or `f`
2609
- (requirement-coverage mismatch) on a `P-Req-*` item. Kinds `b`/`c`/`e` are
2610
- plan-prose defects: they degrade the document, not the resulting code.
2911
+ unsafe code — the single-vote-blocking kind `a` (cited path/symbol mismatch)
2912
+ on any item but `P-Var-*`, or `f` (requirement-coverage mismatch) on a
2913
+ `P-Req-*` item.
2914
+ Kinds `b`/`c`/`e` are plan-prose defects: they degrade the document, not the
2915
+ resulting code. Rollback ordering (`d`) is advisory — a human runs the
2916
+ rollback — so it never counts as correctness-critical. A `P-Var-*` item is
2917
+ majority-gated end to end, so a kind-`a` dissent on one is no more critical
2918
+ than the `b`/`e` its defect should have been raised under; otherwise the same
2919
+ mis-tag that no longer single-vote-blocks would still veto the downgrade.
2611
2920
  """
2921
+ if _is_variation_point_item(item):
2922
+ return False
2612
2923
  kinds = _disagree_breakage_kinds(item)
2613
2924
  is_req = str(item.get("id") or "").upper().startswith("P-REQ")
2614
2925
  return bool(kinds & _SINGLE_VOTE_BLOCKING_KINDS) or (is_req and "f" in kinds)
@@ -3677,20 +3988,6 @@ def _validate_missing_qa_categories_recorded(
3677
3988
  )
3678
3989
 
3679
3990
 
3680
- def _load_report_data(report_path: Path) -> dict | None:
3681
- """The report's data.json, or ``None`` when it is missing or malformed.
3682
- Both cases are already reported by `validate_final_report_data`; callers
3683
- here only need to skip rather than re-report."""
3684
- data_path = _data_path_for(report_path)
3685
- if not data_path.is_file():
3686
- return None
3687
- try:
3688
- loaded = json.loads(data_path.read_text(encoding="utf-8"))
3689
- except (OSError, json.JSONDecodeError):
3690
- return None
3691
- return loaded if isinstance(loaded, dict) else None
3692
-
3693
-
3694
3991
  def _validate_verification_target_match(
3695
3992
  data: dict,
3696
3993
  run_manifest: dict,
@@ -4131,6 +4428,87 @@ def _validate_plan_item_extraction_completeness(
4131
4428
  )
4132
4429
 
4133
4430
 
4431
+ def _validate_variation_point_analysis(
4432
+ vpa: object,
4433
+ architecture_style: str,
4434
+ failures: list[str],
4435
+ ) -> None:
4436
+ """Conditional rules + architecture-style overlay for variation points.
4437
+
4438
+ The schema enforces shape only. Two layers of meaning sit on top:
4439
+
4440
+ Layer 1 is style-agnostic. Declaring "no variation exists" is a claim that
4441
+ needs a written reason, and it must not be paired with declared points —
4442
+ `plan_items._extract_variation_point_items` emits a lone `P-Var-0` in that
4443
+ branch and drops them, so the contradiction would silently exempt every
4444
+ declared point from per-point verification. `extract: true` is a claim in
4445
+ the same way: it names the interface the next implementation plugs into and
4446
+ the Stage Map stage that builds it, so both fields have to be filled. The
4447
+ schema cannot carry this as a `minLength` — the empty string is the natural
4448
+ shape of an `extract: false` decision.
4449
+
4450
+ Layer 2 fires only for a project that declares `architecture.style`
4451
+ `hexagonal`: extracting a variation point there means introducing a port,
4452
+ not a helper. An unconfigured project resolves to `none` and keeps layer-1
4453
+ behaviour only.
4454
+ """
4455
+ if not isinstance(vpa, dict):
4456
+ failures.append("variationPointAnalysis is missing or not an object")
4457
+ return
4458
+ # Schema violations are reported, not raised, so this check still runs on a
4459
+ # malformed block. A type guard here keeps a bad field from aborting the
4460
+ # whole validation and discarding every failure collected so far.
4461
+ raw_points = vpa.get("points")
4462
+ if raw_points is not None and not isinstance(raw_points, list):
4463
+ failures.append("variationPointAnalysis: points must be an array")
4464
+ return
4465
+ points = raw_points or []
4466
+ if not bool(vpa.get("hasMultipleImplementations")):
4467
+ rationale = vpa.get("noVariationRationale")
4468
+ if not isinstance(rationale, str) or not rationale.strip():
4469
+ failures.append(
4470
+ "variationPointAnalysis: hasMultipleImplementations=false "
4471
+ "requires a non-empty noVariationRationale"
4472
+ )
4473
+ if points:
4474
+ failures.append(
4475
+ "variationPointAnalysis: hasMultipleImplementations=false but "
4476
+ "points is non-empty — declared variation points would be "
4477
+ "silently dropped"
4478
+ )
4479
+ return
4480
+ if not points:
4481
+ failures.append(
4482
+ "variationPointAnalysis: hasMultipleImplementations=true requires "
4483
+ "at least one point"
4484
+ )
4485
+ return
4486
+ for index, point in enumerate(points, start=1):
4487
+ if not isinstance(point, dict):
4488
+ failures.append(
4489
+ f"variationPointAnalysis point {index}: must be an object"
4490
+ )
4491
+ continue
4492
+ decision = point.get("extractionDecision")
4493
+ if not isinstance(decision, dict):
4494
+ continue # shape is the schema's job; don't double-report it.
4495
+ if not decision.get("extract"):
4496
+ continue
4497
+ for field in ("interfaceKind", "coveredBy"):
4498
+ value = decision.get(field)
4499
+ if not isinstance(value, str) or not value.strip():
4500
+ failures.append(
4501
+ f"variationPointAnalysis point {index}: extract=true "
4502
+ f"requires a non-empty {field}, got {value!r}"
4503
+ )
4504
+ if architecture_style == "hexagonal" and decision.get("interfaceKind") != "port":
4505
+ failures.append(
4506
+ f"variationPointAnalysis point {index}: architecture style "
4507
+ f"'hexagonal' requires interfaceKind 'port', got "
4508
+ f"{decision.get('interfaceKind')!r}"
4509
+ )
4510
+
4511
+
4134
4512
  _DESIGN_PREP_CONTRACT = "implementation-design-prep-v1"
4135
4513
  _DESIGN_PREP_REQUEST_STATUSES = {"provisional", "blocked"}
4136
4514
  _DESIGN_PREP_TERMINAL_STATUSES = {"ready", "not-applicable"}
@@ -6209,20 +6587,20 @@ def main() -> int:
6209
6587
  report_contracts = _normalize_report_contracts(
6210
6588
  run_manifest.get("reportContracts")
6211
6589
  )
6212
- validate_final_report_data(
6590
+ report_data = validate_final_report_data(
6213
6591
  report_path,
6214
6592
  failures,
6215
6593
  report_contracts=report_contracts,
6216
6594
  )
6217
- _validate_fix_cycle(
6218
- run_manifest, _load_final_report_data(report_path), failures
6219
- )
6595
+ validation_data = report_data if isinstance(report_data, Mapping) else {}
6596
+ _validate_fix_cycle(run_manifest, validation_data, failures)
6220
6597
  validate_report(
6221
6598
  report_path,
6222
6599
  contract["required_agent_status_entries"],
6223
6600
  failures,
6224
6601
  allow_unavailable_token_usage=_session_accounting(team_state) == "artifact-only",
6225
6602
  unavailable_ok_labels=_unavailable_usage_worker_labels(team_state),
6603
+ report_data=validation_data,
6226
6604
  )
6227
6605
  validate_team_state_usage(team_state, failures)
6228
6606
 
@@ -6270,11 +6648,12 @@ def main() -> int:
6270
6648
  if task_type in _END_STATE_PHASES:
6271
6649
  if task_type == "implementation-planning":
6272
6650
  _validate_planning_conformance_declared(report_path, failures)
6273
- report_data = _load_final_report_data(report_path)
6274
- _validate_end_state_coverage(report_data, brief_path, failures)
6651
+ _validate_end_state_coverage(validation_data, brief_path, failures)
6275
6652
  if task_type == "implementation-planning":
6276
- _validate_requirement_provenance(report_data, brief_path, failures)
6277
- _validate_stage_has_requirement(report_data, failures)
6653
+ _validate_requirement_provenance(
6654
+ validation_data, brief_path, failures
6655
+ )
6656
+ _validate_stage_has_requirement(validation_data, failures)
6278
6657
  if task_type == "improvement-discovery":
6279
6658
  run_dir = report_path.parent.parent
6280
6659
  _validate_improvement_discovery(report_path, run_dir, brief_path, failures)
@@ -6291,22 +6670,20 @@ def main() -> int:
6291
6670
  validate_report_views(report_path, failures)
6292
6671
  if task_type == "implementation":
6293
6672
  _validate_missing_qa_categories_recorded(
6294
- _load_report_data(report_path) or {}, project_root, failures
6673
+ validation_data, project_root, failures
6295
6674
  )
6296
6675
  if task_type == "implementation-planning":
6297
- _validate_plan_body_state_file(
6298
- _load_report_data(report_path) or {}, report_path, failures
6299
- )
6676
+ _validate_plan_body_state_file(validation_data, report_path, failures)
6300
6677
  if task_type == "final-verification":
6301
6678
  _validate_verification_target_match(
6302
- _load_report_data(report_path) or {},
6679
+ validation_data,
6303
6680
  run_manifest,
6304
6681
  project_root,
6305
6682
  failures,
6306
6683
  )
6307
6684
  # Phase-agnostic: any task-type can be launched with a carry-in.
6308
6685
  _validate_clarification_carry_in_recorded(
6309
- _load_report_data(report_path) or {},
6686
+ validation_data,
6310
6687
  report_path,
6311
6688
  run_manifest_path,
6312
6689
  project_root,
@@ -6315,7 +6692,12 @@ def main() -> int:
6315
6692
 
6316
6693
  validation_status = "passed" if not failures else "failed"
6317
6694
  update_validation_metadata(
6318
- team_state, run_manifest, task_manifest, validation_status, failures
6695
+ team_state,
6696
+ run_manifest,
6697
+ task_manifest,
6698
+ validation_status,
6699
+ failures,
6700
+ report_data=validation_data,
6319
6701
  )
6320
6702
 
6321
6703
  write_json(team_state_path, team_state)