okstra 0.141.3 → 0.143.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +11 -2
- package/docs/cli.md +15 -0
- package/docs/for-ai/skills/okstra-setup.md +8 -0
- package/docs/project-structure-overview.md +6 -0
- package/docs/task-process/error-analysis.md +9 -4
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +4 -2
- package/runtime/prompts/coding-preflight/architectures/hexagonal.md +3 -3
- package/runtime/prompts/coding-preflight/overview.md +1 -1
- package/runtime/prompts/lead/adapters/claude-code.md +2 -2
- package/runtime/prompts/lead/context-loader.md +2 -2
- package/runtime/prompts/lead/convergence.md +5 -2
- package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
- package/runtime/prompts/lead/plan-body-verification.md +20 -9
- package/runtime/prompts/lead/report-writer.md +4 -3
- package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -0
- package/runtime/prompts/profiles/_common-contract.md +3 -1
- package/runtime/prompts/profiles/_implementation-deliverable.md +3 -3
- package/runtime/prompts/profiles/_implementation-diff-review.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +2 -1
- package/runtime/prompts/profiles/error-analysis.md +5 -1
- package/runtime/prompts/profiles/forbidden-actions.json +0 -1
- package/runtime/prompts/profiles/implementation-planning.md +7 -2
- package/runtime/prompts/profiles/requirements-discovery.md +7 -0
- package/runtime/python/okstra_ctl/analysis_packet.py +28 -3
- package/runtime/python/okstra_ctl/brief_frontmatter.py +56 -0
- package/runtime/python/okstra_ctl/clarification_items.py +99 -5
- package/runtime/python/okstra_ctl/convergence_engine.py +66 -15
- package/runtime/python/okstra_ctl/paths.py +34 -7
- package/runtime/python/okstra_ctl/phase_cleanup.py +235 -0
- package/runtime/python/okstra_ctl/plan_items.py +38 -0
- package/runtime/python/okstra_ctl/run.py +81 -33
- package/runtime/python/okstra_ctl/schema_excerpt.py +5 -3
- package/runtime/python/okstra_ctl/wizard.py +18 -44
- package/runtime/python/okstra_ctl/worker_heartbeat.py +15 -5
- package/runtime/python/okstra_ctl/workflow.py +1 -1
- package/runtime/python/okstra_project/resolver.py +25 -0
- package/runtime/schemas/final-report-v1.0.schema.json +162 -4
- package/runtime/skills/okstra-run/SKILL.md +3 -1
- package/runtime/skills/okstra-setup/SKILL.md +3 -0
- package/runtime/skills/okstra-setup/references/project-config.md +47 -0
- package/runtime/templates/reports/final-report.template.md +51 -0
- package/runtime/templates/reports/i18n/en.json +32 -3
- package/runtime/templates/reports/i18n/ko.json +32 -3
- package/runtime/templates/reports/implementation-input.template.md +1 -2
- package/runtime/templates/reports/task-brief.template.md +1 -1
- package/runtime/validators/validate-brief.py +5 -1
- package/runtime/validators/validate-run.py +430 -48
- package/src/cli-registry.mjs +10 -0
- package/src/commands/execute/phase-cleanup.mjs +38 -0
|
@@ -8,9 +8,12 @@ import importlib.util
|
|
|
8
8
|
import json
|
|
9
9
|
import os
|
|
10
10
|
import re
|
|
11
|
+
import shlex
|
|
11
12
|
import sys
|
|
13
|
+
from collections.abc import Mapping
|
|
12
14
|
from datetime import datetime, timezone
|
|
13
15
|
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
14
17
|
|
|
15
18
|
# Make the okstra packages importable however this validator is reached:
|
|
16
19
|
# ``scripts/`` next to the repo checkout, ``python/`` next to the installed
|
|
@@ -34,6 +37,7 @@ except ImportError: # pragma: no cover — runtime guarantees this import
|
|
|
34
37
|
|
|
35
38
|
from okstra_project import project_json_path # noqa: E402
|
|
36
39
|
from okstra_project.dirs import tasks_root as _okstra_tasks_root # noqa: E402
|
|
40
|
+
from okstra_project.resolver import resolve_architecture # noqa: E402
|
|
37
41
|
|
|
38
42
|
from okstra_ctl.conformance import ( # noqa: E402
|
|
39
43
|
CAPABILITY_WHITELIST,
|
|
@@ -187,10 +191,29 @@ def advance_next_phase(
|
|
|
187
191
|
return default_next_phase(current_phase)
|
|
188
192
|
|
|
189
193
|
|
|
194
|
+
def _error_analysis_next_phase(data: Mapping[str, Any]) -> str | None:
|
|
195
|
+
if not isinstance(data, Mapping):
|
|
196
|
+
return None
|
|
197
|
+
header = data.get("header")
|
|
198
|
+
if not isinstance(header, Mapping) or header.get("taskType") != "error-analysis":
|
|
199
|
+
return None
|
|
200
|
+
error_analysis = data.get("errorAnalysis")
|
|
201
|
+
if not isinstance(error_analysis, Mapping):
|
|
202
|
+
return None
|
|
203
|
+
routing = error_analysis.get("routing")
|
|
204
|
+
if not isinstance(routing, Mapping):
|
|
205
|
+
return None
|
|
206
|
+
target = routing.get("nextTaskType")
|
|
207
|
+
if target in {"error-analysis", "implementation-planning"}:
|
|
208
|
+
return str(target)
|
|
209
|
+
return None
|
|
210
|
+
|
|
211
|
+
|
|
190
212
|
def update_workflow_metadata(
|
|
191
213
|
run_manifest: dict,
|
|
192
214
|
task_manifest: dict,
|
|
193
215
|
validation_status: str,
|
|
216
|
+
report_data: Mapping[str, Any] | None = None,
|
|
194
217
|
) -> None:
|
|
195
218
|
workflow = task_manifest.get("workflow", {})
|
|
196
219
|
if not isinstance(workflow, dict):
|
|
@@ -220,7 +243,14 @@ def update_workflow_metadata(
|
|
|
220
243
|
# Validation just passed → actively advance to the next phase in
|
|
221
244
|
# the sequence rather than preserving a stale value that may equal
|
|
222
245
|
# current_phase (which would cause the lifecycle pointer to stall).
|
|
223
|
-
|
|
246
|
+
report_next_phase = (
|
|
247
|
+
_error_analysis_next_phase(report_data or {})
|
|
248
|
+
if current_phase == "error-analysis"
|
|
249
|
+
else None
|
|
250
|
+
)
|
|
251
|
+
next_recommended_phase = report_next_phase or advance_next_phase(
|
|
252
|
+
current_phase, phase_sequence
|
|
253
|
+
)
|
|
224
254
|
else:
|
|
225
255
|
current_phase_state = "blocked"
|
|
226
256
|
if current_phase:
|
|
@@ -321,6 +351,7 @@ def update_validation_metadata(
|
|
|
321
351
|
task_manifest: dict,
|
|
322
352
|
validation_status: str,
|
|
323
353
|
failures: list[str],
|
|
354
|
+
report_data: Mapping[str, Any] | None = None,
|
|
324
355
|
) -> None:
|
|
325
356
|
checked_at = utc_now()
|
|
326
357
|
|
|
@@ -349,7 +380,12 @@ def update_validation_metadata(
|
|
|
349
380
|
task_manifest["currentStatus"] = (
|
|
350
381
|
"completed" if validation_status == "passed" else "contract-violated"
|
|
351
382
|
)
|
|
352
|
-
update_workflow_metadata(
|
|
383
|
+
update_workflow_metadata(
|
|
384
|
+
run_manifest,
|
|
385
|
+
task_manifest,
|
|
386
|
+
validation_status,
|
|
387
|
+
report_data=report_data,
|
|
388
|
+
)
|
|
353
389
|
|
|
354
390
|
|
|
355
391
|
def extract_contract(
|
|
@@ -1644,7 +1680,10 @@ def _validate_conformance(
|
|
|
1644
1680
|
|
|
1645
1681
|
|
|
1646
1682
|
def _check_incremental_audit_block(
|
|
1647
|
-
report_path: Path,
|
|
1683
|
+
report_path: Path,
|
|
1684
|
+
content: str,
|
|
1685
|
+
failures: list[str],
|
|
1686
|
+
report_data: Mapping[str, Any] | None = None,
|
|
1648
1687
|
) -> None:
|
|
1649
1688
|
"""Enforce the Section 0 incremental audit block.
|
|
1650
1689
|
|
|
@@ -1656,7 +1695,11 @@ def _check_incremental_audit_block(
|
|
|
1656
1695
|
decision are exempt, so this never conflicts with the empty-Section-0
|
|
1657
1696
|
stub rule (that fires only when NO carry-in was provided at all).
|
|
1658
1697
|
"""
|
|
1659
|
-
data =
|
|
1698
|
+
data = (
|
|
1699
|
+
report_data
|
|
1700
|
+
if report_data is not None
|
|
1701
|
+
else _load_final_report_data(report_path)
|
|
1702
|
+
)
|
|
1660
1703
|
planning = data.get("implementationPlanning")
|
|
1661
1704
|
if not isinstance(planning, dict):
|
|
1662
1705
|
return
|
|
@@ -1688,6 +1731,7 @@ def validate_report(
|
|
|
1688
1731
|
*,
|
|
1689
1732
|
allow_unavailable_token_usage: bool = False,
|
|
1690
1733
|
unavailable_ok_labels: frozenset[str] = frozenset(),
|
|
1734
|
+
report_data: Mapping[str, Any] | None = None,
|
|
1691
1735
|
) -> None:
|
|
1692
1736
|
if not report_path.exists():
|
|
1693
1737
|
failures.append(f"final report is missing: {report_path}")
|
|
@@ -1778,7 +1822,12 @@ def validate_report(
|
|
|
1778
1822
|
|
|
1779
1823
|
# Incremental audit block — an `incremental`-mode re-run must expose its
|
|
1780
1824
|
# scope decision in Section 0 so the narrowed re-verification is auditable.
|
|
1781
|
-
_check_incremental_audit_block(
|
|
1825
|
+
_check_incremental_audit_block(
|
|
1826
|
+
report_path,
|
|
1827
|
+
content,
|
|
1828
|
+
failures,
|
|
1829
|
+
report_data=report_data,
|
|
1830
|
+
)
|
|
1782
1831
|
|
|
1783
1832
|
# Deprecated section headings — pre-1.0 hard removal.
|
|
1784
1833
|
for pattern, remedy in _DEPRECATED_FINAL_REPORT_PATTERNS:
|
|
@@ -2239,6 +2288,213 @@ def _validate_verdict_card_fields(data: dict, failures: list[str]) -> None:
|
|
|
2239
2288
|
)
|
|
2240
2289
|
|
|
2241
2290
|
|
|
2291
|
+
def _route_target_matches(value: Any, target: str, *, command: bool) -> bool:
|
|
2292
|
+
if not isinstance(value, str):
|
|
2293
|
+
return False
|
|
2294
|
+
if not command:
|
|
2295
|
+
pattern = rf"(?<![\w-]){re.escape(target)}(?![\w-])"
|
|
2296
|
+
return re.search(pattern, value) is not None
|
|
2297
|
+
try:
|
|
2298
|
+
tokens = shlex.split(value)
|
|
2299
|
+
except ValueError:
|
|
2300
|
+
return False
|
|
2301
|
+
task_type_values: list[str] = []
|
|
2302
|
+
for index, token in enumerate(tokens):
|
|
2303
|
+
if token in {"task-type", "--task-type"}:
|
|
2304
|
+
task_type_values.append(
|
|
2305
|
+
tokens[index + 1] if index + 1 < len(tokens) else ""
|
|
2306
|
+
)
|
|
2307
|
+
for prefix in ("task-type=", "--task-type="):
|
|
2308
|
+
if token.startswith(prefix):
|
|
2309
|
+
task_type_values.append(token[len(prefix):])
|
|
2310
|
+
return bool(task_type_values) and all(
|
|
2311
|
+
task_type_value == target for task_type_value in task_type_values
|
|
2312
|
+
)
|
|
2313
|
+
|
|
2314
|
+
|
|
2315
|
+
def _validate_error_analysis_consistency(
|
|
2316
|
+
data: Mapping[str, Any], failures: list[str]
|
|
2317
|
+
) -> None:
|
|
2318
|
+
error_analysis_value = data.get("errorAnalysis")
|
|
2319
|
+
error_analysis = (
|
|
2320
|
+
error_analysis_value if isinstance(error_analysis_value, Mapping) else {}
|
|
2321
|
+
)
|
|
2322
|
+
reproduction_value = error_analysis.get("reproduction")
|
|
2323
|
+
reproduction = (
|
|
2324
|
+
reproduction_value if isinstance(reproduction_value, Mapping) else {}
|
|
2325
|
+
)
|
|
2326
|
+
reproduction_status = reproduction.get("status")
|
|
2327
|
+
blocked_reason = reproduction.get("blockedReason")
|
|
2328
|
+
if reproduction_status == "blocked-before-repro":
|
|
2329
|
+
if not isinstance(blocked_reason, str) or not blocked_reason.strip():
|
|
2330
|
+
failures.append(
|
|
2331
|
+
"final-report data.json: blocked-before-repro requires a non-empty "
|
|
2332
|
+
"errorAnalysis.reproduction.blockedReason."
|
|
2333
|
+
)
|
|
2334
|
+
elif blocked_reason != "":
|
|
2335
|
+
failures.append(
|
|
2336
|
+
"final-report data.json: errorAnalysis.reproduction.blockedReason "
|
|
2337
|
+
"must be exactly empty unless status is blocked-before-repro."
|
|
2338
|
+
)
|
|
2339
|
+
|
|
2340
|
+
candidates_value = error_analysis.get("causeCandidates")
|
|
2341
|
+
candidates = candidates_value if isinstance(candidates_value, list) else []
|
|
2342
|
+
candidate_ids: list[str] = []
|
|
2343
|
+
for candidate in candidates:
|
|
2344
|
+
if not isinstance(candidate, Mapping):
|
|
2345
|
+
continue
|
|
2346
|
+
candidate_id = candidate.get("id")
|
|
2347
|
+
if isinstance(candidate_id, str):
|
|
2348
|
+
candidate_ids.append(candidate_id)
|
|
2349
|
+
duplicate_ids = sorted(
|
|
2350
|
+
candidate_id
|
|
2351
|
+
for candidate_id in set(candidate_ids)
|
|
2352
|
+
if candidate_ids.count(candidate_id) > 1
|
|
2353
|
+
)
|
|
2354
|
+
if duplicate_ids:
|
|
2355
|
+
failures.append(
|
|
2356
|
+
"final-report data.json: duplicate cause candidate id(s): "
|
|
2357
|
+
+ ", ".join(duplicate_ids)
|
|
2358
|
+
+ "."
|
|
2359
|
+
)
|
|
2360
|
+
|
|
2361
|
+
routing_value = error_analysis.get("routing")
|
|
2362
|
+
routing = routing_value if isinstance(routing_value, Mapping) else {}
|
|
2363
|
+
target = routing.get("nextTaskType")
|
|
2364
|
+
leading_cause_id = routing.get("leadingCauseId")
|
|
2365
|
+
candidate_id_set = set(candidate_ids)
|
|
2366
|
+
if target == "implementation-planning":
|
|
2367
|
+
if not candidates:
|
|
2368
|
+
failures.append(
|
|
2369
|
+
"final-report data.json: implementation-planning routing requires "
|
|
2370
|
+
"at least one cause candidate."
|
|
2371
|
+
)
|
|
2372
|
+
if (
|
|
2373
|
+
not isinstance(leading_cause_id, str)
|
|
2374
|
+
or leading_cause_id not in candidate_id_set
|
|
2375
|
+
):
|
|
2376
|
+
failures.append(
|
|
2377
|
+
"final-report data.json: implementation-planning routing "
|
|
2378
|
+
"leadingCauseId must reference a cause candidate."
|
|
2379
|
+
)
|
|
2380
|
+
elif target == "error-analysis" and (
|
|
2381
|
+
not isinstance(leading_cause_id, str)
|
|
2382
|
+
or (leading_cause_id != "" and leading_cause_id not in candidate_id_set)
|
|
2383
|
+
):
|
|
2384
|
+
failures.append(
|
|
2385
|
+
"final-report data.json: error-analysis routing leadingCauseId must be "
|
|
2386
|
+
"empty or reference a cause candidate."
|
|
2387
|
+
)
|
|
2388
|
+
|
|
2389
|
+
expected_direction = {
|
|
2390
|
+
"implementation-planning": "begin-planning",
|
|
2391
|
+
"error-analysis": "continue-investigation",
|
|
2392
|
+
}.get(target)
|
|
2393
|
+
verdict_card_value = data.get("verdictCard")
|
|
2394
|
+
verdict_card = (
|
|
2395
|
+
verdict_card_value if isinstance(verdict_card_value, Mapping) else {}
|
|
2396
|
+
)
|
|
2397
|
+
final_verdict_value = data.get("finalVerdict")
|
|
2398
|
+
final_verdict = (
|
|
2399
|
+
final_verdict_value if isinstance(final_verdict_value, Mapping) else {}
|
|
2400
|
+
)
|
|
2401
|
+
if expected_direction:
|
|
2402
|
+
for field_name, verdict in (
|
|
2403
|
+
("verdictCard", verdict_card),
|
|
2404
|
+
("finalVerdict", final_verdict),
|
|
2405
|
+
):
|
|
2406
|
+
if verdict.get("direction") != expected_direction:
|
|
2407
|
+
failures.append(
|
|
2408
|
+
f"final-report data.json: {field_name}.direction must be "
|
|
2409
|
+
f"`{expected_direction}` for `{target}` routing."
|
|
2410
|
+
)
|
|
2411
|
+
|
|
2412
|
+
follow_up_tasks_value = data.get("followUpTasks")
|
|
2413
|
+
follow_up_tasks = (
|
|
2414
|
+
follow_up_tasks_value if isinstance(follow_up_tasks_value, list) else []
|
|
2415
|
+
)
|
|
2416
|
+
continuations = [
|
|
2417
|
+
row
|
|
2418
|
+
for row in follow_up_tasks
|
|
2419
|
+
if isinstance(row, Mapping) and row.get("origin") == "phase-continuation"
|
|
2420
|
+
]
|
|
2421
|
+
if len(continuations) != 1:
|
|
2422
|
+
failures.append(
|
|
2423
|
+
"final-report data.json: followUpTasks must contain exactly one "
|
|
2424
|
+
"phase-continuation row."
|
|
2425
|
+
)
|
|
2426
|
+
else:
|
|
2427
|
+
continuation = continuations[0]
|
|
2428
|
+
frontmatter_value = data.get("frontmatter")
|
|
2429
|
+
frontmatter = (
|
|
2430
|
+
frontmatter_value if isinstance(frontmatter_value, Mapping) else {}
|
|
2431
|
+
)
|
|
2432
|
+
task_id = frontmatter.get("taskId")
|
|
2433
|
+
if continuation.get("suggestedTaskType") != target:
|
|
2434
|
+
failures.append(
|
|
2435
|
+
"final-report data.json: phase-continuation suggestedTaskType "
|
|
2436
|
+
"must match the routing target."
|
|
2437
|
+
)
|
|
2438
|
+
if continuation.get("newTaskId") != task_id:
|
|
2439
|
+
failures.append(
|
|
2440
|
+
"final-report data.json: phase-continuation newTaskId must match "
|
|
2441
|
+
"frontmatter.taskId."
|
|
2442
|
+
)
|
|
2443
|
+
if continuation.get("priority") != "P0":
|
|
2444
|
+
failures.append(
|
|
2445
|
+
"final-report data.json: phase-continuation priority must be P0."
|
|
2446
|
+
)
|
|
2447
|
+
if continuation.get("autoSpawn") != "no":
|
|
2448
|
+
failures.append(
|
|
2449
|
+
"final-report data.json: phase-continuation autoSpawn must be no."
|
|
2450
|
+
)
|
|
2451
|
+
|
|
2452
|
+
if isinstance(target, str) and target in {
|
|
2453
|
+
"error-analysis",
|
|
2454
|
+
"implementation-planning",
|
|
2455
|
+
}:
|
|
2456
|
+
for field_name, value in (
|
|
2457
|
+
("verdictCard.nextStep", verdict_card.get("nextStep")),
|
|
2458
|
+
("finalVerdict.nextStep", final_verdict.get("nextStep")),
|
|
2459
|
+
):
|
|
2460
|
+
if not _route_target_matches(value, target, command=False):
|
|
2461
|
+
failures.append(
|
|
2462
|
+
f"final-report data.json: {field_name} must contain routing "
|
|
2463
|
+
f"target `{target}`."
|
|
2464
|
+
)
|
|
2465
|
+
|
|
2466
|
+
next_steps_value = data.get("recommendedNextSteps")
|
|
2467
|
+
next_steps = next_steps_value if isinstance(next_steps_value, list) else []
|
|
2468
|
+
first_step = (
|
|
2469
|
+
next_steps[0]
|
|
2470
|
+
if next_steps and isinstance(next_steps[0], Mapping)
|
|
2471
|
+
else {}
|
|
2472
|
+
)
|
|
2473
|
+
step_text = first_step.get("text")
|
|
2474
|
+
if not _route_target_matches(step_text, target, command=False):
|
|
2475
|
+
failures.append(
|
|
2476
|
+
"final-report data.json: recommendedNextSteps[0].text must contain "
|
|
2477
|
+
f"routing target `{target}`."
|
|
2478
|
+
)
|
|
2479
|
+
commands_value = first_step.get("commands")
|
|
2480
|
+
commands = commands_value if isinstance(commands_value, list) else []
|
|
2481
|
+
if not commands:
|
|
2482
|
+
failures.append(
|
|
2483
|
+
"final-report data.json: recommendedNextSteps[0].commands must "
|
|
2484
|
+
"contain at least one command."
|
|
2485
|
+
)
|
|
2486
|
+
for index, command_value in enumerate(commands):
|
|
2487
|
+
command = command_value if isinstance(command_value, Mapping) else {}
|
|
2488
|
+
for command_field in ("claudeCode", "terminal"):
|
|
2489
|
+
value = command.get(command_field)
|
|
2490
|
+
if not _route_target_matches(value, target, command=True):
|
|
2491
|
+
failures.append(
|
|
2492
|
+
"final-report data.json: recommendedNextSteps[0].commands"
|
|
2493
|
+
f"[{index}].{command_field} must contain routing target "
|
|
2494
|
+
f"`{target}`."
|
|
2495
|
+
)
|
|
2496
|
+
|
|
2497
|
+
|
|
2242
2498
|
def _load_final_report_data(report_path: Path) -> dict:
|
|
2243
2499
|
"""Best-effort parse of the final-report data.json sibling. Returns {} when
|
|
2244
2500
|
absent or unparseable — those conditions are already surfaced as failures by
|
|
@@ -2257,7 +2513,7 @@ def validate_final_report_data(
|
|
|
2257
2513
|
failures: list[str],
|
|
2258
2514
|
*,
|
|
2259
2515
|
report_contracts: set[str] | None = None,
|
|
2260
|
-
) -> None:
|
|
2516
|
+
) -> Mapping[str, Any] | None:
|
|
2261
2517
|
"""Validate the final-report data.json against the v1.0 schema.
|
|
2262
2518
|
|
|
2263
2519
|
The data.json is the source-of-truth that the renderer reads to
|
|
@@ -2270,6 +2526,8 @@ def validate_final_report_data(
|
|
|
2270
2526
|
Missing data.json is reported as a single failure rather than a
|
|
2271
2527
|
cascade of substring failures — that points the writer at the right
|
|
2272
2528
|
fix (write the data.json) instead of futilely editing the markdown.
|
|
2529
|
+
The returned mapping is the exact loaded snapshot consumed by later
|
|
2530
|
+
finalization checks and workflow persistence.
|
|
2273
2531
|
"""
|
|
2274
2532
|
if schema_validate is None or load_schema is None:
|
|
2275
2533
|
# Module-load fallback path; should never fire in a real install.
|
|
@@ -2314,7 +2572,9 @@ def validate_final_report_data(
|
|
|
2314
2572
|
_validate_verifier_fail_blocks_verdict(data, failures)
|
|
2315
2573
|
if task_type == "implementation":
|
|
2316
2574
|
_validate_stage_carry_sidecar_exists(data, report_path, failures)
|
|
2317
|
-
if task_type == "
|
|
2575
|
+
if task_type == "error-analysis":
|
|
2576
|
+
_validate_error_analysis_consistency(data, failures)
|
|
2577
|
+
elif task_type == "final-verification":
|
|
2318
2578
|
_validate_final_verification_consistency(data, failures)
|
|
2319
2579
|
_validate_verified_row_recorded(data, report_path, failures)
|
|
2320
2580
|
elif task_type == "implementation-planning":
|
|
@@ -2343,6 +2603,11 @@ def validate_final_report_data(
|
|
|
2343
2603
|
_validate_round_recorded_verdicts(data, failures)
|
|
2344
2604
|
_validate_verdicts_match_current_subjects(data, failures)
|
|
2345
2605
|
_validate_plan_item_extraction_completeness(data, failures)
|
|
2606
|
+
_validate_variation_point_analysis(
|
|
2607
|
+
(data.get("implementationPlanning") or {}).get("variationPointAnalysis"),
|
|
2608
|
+
resolve_architecture(_project_root_from_report(report_path)),
|
|
2609
|
+
failures,
|
|
2610
|
+
)
|
|
2346
2611
|
_validate_plan_item_subject_substance(data, failures)
|
|
2347
2612
|
_validate_plan_body_clarification_matching(data, failures)
|
|
2348
2613
|
_validate_disagree_has_fixability(data, failures)
|
|
@@ -2358,6 +2623,8 @@ def validate_final_report_data(
|
|
|
2358
2623
|
for warning in warnings:
|
|
2359
2624
|
print(f"validate-run: warning: {warning}", file=sys.stderr)
|
|
2360
2625
|
|
|
2626
|
+
return data
|
|
2627
|
+
|
|
2361
2628
|
|
|
2362
2629
|
# path:line (foo.service.ts:268), an in-report ID (C-001 / E-006 / P-004), a
|
|
2363
2630
|
# namespaced audit ref (claude:F-005), or a §section reference (§5.4) — any one
|
|
@@ -2524,10 +2791,16 @@ _PLAN_GATE_RANK = {
|
|
|
2524
2791
|
|
|
2525
2792
|
# Breakage kinds where a single DISAGREE blocks the gate on its own (no majority
|
|
2526
2793
|
# needed), because the defect is concrete, safety-critical, and adversarially
|
|
2527
|
-
# verifiable: `a` = cited path/symbol mismatch
|
|
2528
|
-
#
|
|
2529
|
-
#
|
|
2530
|
-
_SINGLE_VOTE_BLOCKING_KINDS = {"a"
|
|
2794
|
+
# verifiable: `a` = cited path/symbol mismatch. `b`/`c`/`e` still need a
|
|
2795
|
+
# majority — `b` in particular is prone to planning-vs-implementation
|
|
2796
|
+
# environment false positives.
|
|
2797
|
+
_SINGLE_VOTE_BLOCKING_KINDS = {"a"}
|
|
2798
|
+
|
|
2799
|
+
# Rollback ordering (`d`) is executed by a human, not by okstra's workers or
|
|
2800
|
+
# verifiers, so a rollback-ordering dissent is recorded but never gates
|
|
2801
|
+
# approval: it is dropped from the blocking-disagree tally entirely, so an item
|
|
2802
|
+
# whose only DISAGREEs are advisory-only can never rise above `has-dissent`.
|
|
2803
|
+
_ADVISORY_ONLY_KINDS = {"d"}
|
|
2531
2804
|
|
|
2532
2805
|
# Stop reasons that justify promoting a still-broken planner-fixable item to the
|
|
2533
2806
|
# user: the self-fix budget ran out, or a round produced no net resolution so
|
|
@@ -2535,6 +2808,20 @@ _SINGLE_VOTE_BLOCKING_KINDS = {"a", "d"}
|
|
|
2535
2808
|
_SELF_FIX_EXHAUSTED_REASONS = frozenset({"max-rounds-reached", "no-progress"})
|
|
2536
2809
|
|
|
2537
2810
|
|
|
2811
|
+
def _is_variation_point_item(item: dict) -> bool:
|
|
2812
|
+
"""Whether this is a `P-Var-*` variation-point item, which is majority-gated
|
|
2813
|
+
(`prompts/lead/plan-body-verification.md` "`P-Var-<N>` … is majority-gated").
|
|
2814
|
+
Whether a behavior has two implementations, and whether the plan extracted the
|
|
2815
|
+
right interface for it, is a design judgement — it lacks the concrete certainty
|
|
2816
|
+
of kind `a`, where a verifier points at two spelled-out references that
|
|
2817
|
+
contradict each other. So kind `a` carries no extra weight on a P-Var item: it
|
|
2818
|
+
neither single-vote-blocks nor counts as correctness-critical, exactly like the
|
|
2819
|
+
`b` / `c` / `e` kinds the prompt routes P-Var defects to. Only a
|
|
2820
|
+
`majority-disagree` gates it — that part is unchanged.
|
|
2821
|
+
"""
|
|
2822
|
+
return str(item.get("id") or "").upper().startswith("P-VAR")
|
|
2823
|
+
|
|
2824
|
+
|
|
2538
2825
|
def _classify_plan_item_gate(item: dict) -> str:
|
|
2539
2826
|
"""Recompute one plan item's gate class from its per-worker verdicts,
|
|
2540
2827
|
per `prompts/lead/plan-body-verification.md` "Round protocol". Returns one of
|
|
@@ -2556,15 +2843,32 @@ def _classify_plan_item_gate(item: dict) -> str:
|
|
|
2556
2843
|
return "all-non-result"
|
|
2557
2844
|
disagree = [(vd, bk) for (vd, bk) in non_error if vd == "DISAGREE"]
|
|
2558
2845
|
agree = [(vd, bk) for (vd, bk) in non_error if vd in ("AGREE", "SUPPLEMENT")]
|
|
2559
|
-
disagree_kinds = {bk for (_vd, bk) in disagree if bk}
|
|
2560
2846
|
if not disagree:
|
|
2561
2847
|
return "full-consensus"
|
|
2848
|
+
# Rollback is a human-run operation, so rollback dissent never blocks the
|
|
2849
|
+
# gate — closed from two angles so a verifier cannot re-block it by relabelling:
|
|
2850
|
+
# (1) a whole rollback plan item (`P-Rb-*`) is advisory regardless of
|
|
2851
|
+
# breakage kind — otherwise a `DISAGREE(b)` "rollback command is
|
|
2852
|
+
# ambiguous" would sail past the kind-`d` exemption and block;
|
|
2853
|
+
# (2) a rollback-ordering dissent (`d`) is advisory on ANY item, since a
|
|
2854
|
+
# rollback-order defect raised against a non-rollback item is still a
|
|
2855
|
+
# human-run concern.
|
|
2856
|
+
# Both are recorded as dissent and fold into `has-dissent`, never blocking.
|
|
2857
|
+
if str(item.get("id") or "").upper().startswith("P-RB"):
|
|
2858
|
+
return "has-dissent"
|
|
2859
|
+
blocking_disagree = [(vd, bk) for (vd, bk) in disagree if bk not in _ADVISORY_ONLY_KINDS]
|
|
2860
|
+
if not blocking_disagree:
|
|
2861
|
+
return "has-dissent"
|
|
2862
|
+
blocking_kinds = {bk for (_vd, bk) in blocking_disagree if bk}
|
|
2562
2863
|
# Single-vote-blocking kinds: one confirmed DISAGREE on a concrete,
|
|
2563
2864
|
# safety-critical, adversarially-verifiable defect is enough to block, even
|
|
2564
2865
|
# in a two-worker roster — a lone correct dissent must not be outvoted here.
|
|
2565
|
-
# `a
|
|
2866
|
+
# `a` for any item except `P-Var-*` (majority-gated, see
|
|
2867
|
+
# `_is_variation_point_item`); `f` only for P-Req items (requirement coverage).
|
|
2566
2868
|
is_req = str(item.get("id") or "").upper().startswith("P-REQ")
|
|
2567
|
-
if
|
|
2869
|
+
if not _is_variation_point_item(item) and (
|
|
2870
|
+
blocking_kinds & _SINGLE_VOTE_BLOCKING_KINDS or (is_req and "f" in blocking_kinds)
|
|
2871
|
+
):
|
|
2568
2872
|
# "One confirmed DISAGREE" presupposes the item was actually
|
|
2569
2873
|
# cross-verified. When the peer returned a non-result nothing confirmed
|
|
2570
2874
|
# the dissent, so blocking here would reproduce the same
|
|
@@ -2577,7 +2881,7 @@ def _classify_plan_item_gate(item: dict) -> str:
|
|
|
2577
2881
|
# two participating votes, so a lone surviving DISAGREE (its peer returned a
|
|
2578
2882
|
# non-result) does NOT block. That fixes the paradox where a worker failure
|
|
2579
2883
|
# made the gate stricter than a healthy roster would.
|
|
2580
|
-
if len(non_error) >= 2 and len(
|
|
2884
|
+
if len(non_error) >= 2 and len(blocking_disagree) > len(agree):
|
|
2581
2885
|
return "majority-disagree"
|
|
2582
2886
|
return "has-dissent"
|
|
2583
2887
|
|
|
@@ -2604,11 +2908,18 @@ def _has_planner_fixable_majority(item: dict) -> bool:
|
|
|
2604
2908
|
|
|
2605
2909
|
def _is_correctness_critical(item: dict) -> bool:
|
|
2606
2910
|
"""Whether this item's defect would make `implementation` produce wrong or
|
|
2607
|
-
unsafe code — the single-vote-blocking
|
|
2608
|
-
|
|
2609
|
-
|
|
2610
|
-
plan-prose defects: they degrade the document, not the
|
|
2911
|
+
unsafe code — the single-vote-blocking kind `a` (cited path/symbol mismatch)
|
|
2912
|
+
on any item but `P-Var-*`, or `f` (requirement-coverage mismatch) on a
|
|
2913
|
+
`P-Req-*` item.
|
|
2914
|
+
Kinds `b`/`c`/`e` are plan-prose defects: they degrade the document, not the
|
|
2915
|
+
resulting code. Rollback ordering (`d`) is advisory — a human runs the
|
|
2916
|
+
rollback — so it never counts as correctness-critical. A `P-Var-*` item is
|
|
2917
|
+
majority-gated end to end, so a kind-`a` dissent on one is no more critical
|
|
2918
|
+
than the `b`/`e` its defect should have been raised under; otherwise the same
|
|
2919
|
+
mis-tag that no longer single-vote-blocks would still veto the downgrade.
|
|
2611
2920
|
"""
|
|
2921
|
+
if _is_variation_point_item(item):
|
|
2922
|
+
return False
|
|
2612
2923
|
kinds = _disagree_breakage_kinds(item)
|
|
2613
2924
|
is_req = str(item.get("id") or "").upper().startswith("P-REQ")
|
|
2614
2925
|
return bool(kinds & _SINGLE_VOTE_BLOCKING_KINDS) or (is_req and "f" in kinds)
|
|
@@ -3677,20 +3988,6 @@ def _validate_missing_qa_categories_recorded(
|
|
|
3677
3988
|
)
|
|
3678
3989
|
|
|
3679
3990
|
|
|
3680
|
-
def _load_report_data(report_path: Path) -> dict | None:
|
|
3681
|
-
"""The report's data.json, or ``None`` when it is missing or malformed.
|
|
3682
|
-
Both cases are already reported by `validate_final_report_data`; callers
|
|
3683
|
-
here only need to skip rather than re-report."""
|
|
3684
|
-
data_path = _data_path_for(report_path)
|
|
3685
|
-
if not data_path.is_file():
|
|
3686
|
-
return None
|
|
3687
|
-
try:
|
|
3688
|
-
loaded = json.loads(data_path.read_text(encoding="utf-8"))
|
|
3689
|
-
except (OSError, json.JSONDecodeError):
|
|
3690
|
-
return None
|
|
3691
|
-
return loaded if isinstance(loaded, dict) else None
|
|
3692
|
-
|
|
3693
|
-
|
|
3694
3991
|
def _validate_verification_target_match(
|
|
3695
3992
|
data: dict,
|
|
3696
3993
|
run_manifest: dict,
|
|
@@ -4131,6 +4428,87 @@ def _validate_plan_item_extraction_completeness(
|
|
|
4131
4428
|
)
|
|
4132
4429
|
|
|
4133
4430
|
|
|
4431
|
+
def _validate_variation_point_analysis(
|
|
4432
|
+
vpa: object,
|
|
4433
|
+
architecture_style: str,
|
|
4434
|
+
failures: list[str],
|
|
4435
|
+
) -> None:
|
|
4436
|
+
"""Conditional rules + architecture-style overlay for variation points.
|
|
4437
|
+
|
|
4438
|
+
The schema enforces shape only. Two layers of meaning sit on top:
|
|
4439
|
+
|
|
4440
|
+
Layer 1 is style-agnostic. Declaring "no variation exists" is a claim that
|
|
4441
|
+
needs a written reason, and it must not be paired with declared points —
|
|
4442
|
+
`plan_items._extract_variation_point_items` emits a lone `P-Var-0` in that
|
|
4443
|
+
branch and drops them, so the contradiction would silently exempt every
|
|
4444
|
+
declared point from per-point verification. `extract: true` is a claim in
|
|
4445
|
+
the same way: it names the interface the next implementation plugs into and
|
|
4446
|
+
the Stage Map stage that builds it, so both fields have to be filled. The
|
|
4447
|
+
schema cannot carry this as a `minLength` — the empty string is the natural
|
|
4448
|
+
shape of an `extract: false` decision.
|
|
4449
|
+
|
|
4450
|
+
Layer 2 fires only for a project that declares `architecture.style`
|
|
4451
|
+
`hexagonal`: extracting a variation point there means introducing a port,
|
|
4452
|
+
not a helper. An unconfigured project resolves to `none` and keeps layer-1
|
|
4453
|
+
behaviour only.
|
|
4454
|
+
"""
|
|
4455
|
+
if not isinstance(vpa, dict):
|
|
4456
|
+
failures.append("variationPointAnalysis is missing or not an object")
|
|
4457
|
+
return
|
|
4458
|
+
# Schema violations are reported, not raised, so this check still runs on a
|
|
4459
|
+
# malformed block. A type guard here keeps a bad field from aborting the
|
|
4460
|
+
# whole validation and discarding every failure collected so far.
|
|
4461
|
+
raw_points = vpa.get("points")
|
|
4462
|
+
if raw_points is not None and not isinstance(raw_points, list):
|
|
4463
|
+
failures.append("variationPointAnalysis: points must be an array")
|
|
4464
|
+
return
|
|
4465
|
+
points = raw_points or []
|
|
4466
|
+
if not bool(vpa.get("hasMultipleImplementations")):
|
|
4467
|
+
rationale = vpa.get("noVariationRationale")
|
|
4468
|
+
if not isinstance(rationale, str) or not rationale.strip():
|
|
4469
|
+
failures.append(
|
|
4470
|
+
"variationPointAnalysis: hasMultipleImplementations=false "
|
|
4471
|
+
"requires a non-empty noVariationRationale"
|
|
4472
|
+
)
|
|
4473
|
+
if points:
|
|
4474
|
+
failures.append(
|
|
4475
|
+
"variationPointAnalysis: hasMultipleImplementations=false but "
|
|
4476
|
+
"points is non-empty — declared variation points would be "
|
|
4477
|
+
"silently dropped"
|
|
4478
|
+
)
|
|
4479
|
+
return
|
|
4480
|
+
if not points:
|
|
4481
|
+
failures.append(
|
|
4482
|
+
"variationPointAnalysis: hasMultipleImplementations=true requires "
|
|
4483
|
+
"at least one point"
|
|
4484
|
+
)
|
|
4485
|
+
return
|
|
4486
|
+
for index, point in enumerate(points, start=1):
|
|
4487
|
+
if not isinstance(point, dict):
|
|
4488
|
+
failures.append(
|
|
4489
|
+
f"variationPointAnalysis point {index}: must be an object"
|
|
4490
|
+
)
|
|
4491
|
+
continue
|
|
4492
|
+
decision = point.get("extractionDecision")
|
|
4493
|
+
if not isinstance(decision, dict):
|
|
4494
|
+
continue # shape is the schema's job; don't double-report it.
|
|
4495
|
+
if not decision.get("extract"):
|
|
4496
|
+
continue
|
|
4497
|
+
for field in ("interfaceKind", "coveredBy"):
|
|
4498
|
+
value = decision.get(field)
|
|
4499
|
+
if not isinstance(value, str) or not value.strip():
|
|
4500
|
+
failures.append(
|
|
4501
|
+
f"variationPointAnalysis point {index}: extract=true "
|
|
4502
|
+
f"requires a non-empty {field}, got {value!r}"
|
|
4503
|
+
)
|
|
4504
|
+
if architecture_style == "hexagonal" and decision.get("interfaceKind") != "port":
|
|
4505
|
+
failures.append(
|
|
4506
|
+
f"variationPointAnalysis point {index}: architecture style "
|
|
4507
|
+
f"'hexagonal' requires interfaceKind 'port', got "
|
|
4508
|
+
f"{decision.get('interfaceKind')!r}"
|
|
4509
|
+
)
|
|
4510
|
+
|
|
4511
|
+
|
|
4134
4512
|
_DESIGN_PREP_CONTRACT = "implementation-design-prep-v1"
|
|
4135
4513
|
_DESIGN_PREP_REQUEST_STATUSES = {"provisional", "blocked"}
|
|
4136
4514
|
_DESIGN_PREP_TERMINAL_STATUSES = {"ready", "not-applicable"}
|
|
@@ -6209,20 +6587,20 @@ def main() -> int:
|
|
|
6209
6587
|
report_contracts = _normalize_report_contracts(
|
|
6210
6588
|
run_manifest.get("reportContracts")
|
|
6211
6589
|
)
|
|
6212
|
-
validate_final_report_data(
|
|
6590
|
+
report_data = validate_final_report_data(
|
|
6213
6591
|
report_path,
|
|
6214
6592
|
failures,
|
|
6215
6593
|
report_contracts=report_contracts,
|
|
6216
6594
|
)
|
|
6217
|
-
|
|
6218
|
-
|
|
6219
|
-
)
|
|
6595
|
+
validation_data = report_data if isinstance(report_data, Mapping) else {}
|
|
6596
|
+
_validate_fix_cycle(run_manifest, validation_data, failures)
|
|
6220
6597
|
validate_report(
|
|
6221
6598
|
report_path,
|
|
6222
6599
|
contract["required_agent_status_entries"],
|
|
6223
6600
|
failures,
|
|
6224
6601
|
allow_unavailable_token_usage=_session_accounting(team_state) == "artifact-only",
|
|
6225
6602
|
unavailable_ok_labels=_unavailable_usage_worker_labels(team_state),
|
|
6603
|
+
report_data=validation_data,
|
|
6226
6604
|
)
|
|
6227
6605
|
validate_team_state_usage(team_state, failures)
|
|
6228
6606
|
|
|
@@ -6270,11 +6648,12 @@ def main() -> int:
|
|
|
6270
6648
|
if task_type in _END_STATE_PHASES:
|
|
6271
6649
|
if task_type == "implementation-planning":
|
|
6272
6650
|
_validate_planning_conformance_declared(report_path, failures)
|
|
6273
|
-
|
|
6274
|
-
_validate_end_state_coverage(report_data, brief_path, failures)
|
|
6651
|
+
_validate_end_state_coverage(validation_data, brief_path, failures)
|
|
6275
6652
|
if task_type == "implementation-planning":
|
|
6276
|
-
_validate_requirement_provenance(
|
|
6277
|
-
|
|
6653
|
+
_validate_requirement_provenance(
|
|
6654
|
+
validation_data, brief_path, failures
|
|
6655
|
+
)
|
|
6656
|
+
_validate_stage_has_requirement(validation_data, failures)
|
|
6278
6657
|
if task_type == "improvement-discovery":
|
|
6279
6658
|
run_dir = report_path.parent.parent
|
|
6280
6659
|
_validate_improvement_discovery(report_path, run_dir, brief_path, failures)
|
|
@@ -6291,22 +6670,20 @@ def main() -> int:
|
|
|
6291
6670
|
validate_report_views(report_path, failures)
|
|
6292
6671
|
if task_type == "implementation":
|
|
6293
6672
|
_validate_missing_qa_categories_recorded(
|
|
6294
|
-
|
|
6673
|
+
validation_data, project_root, failures
|
|
6295
6674
|
)
|
|
6296
6675
|
if task_type == "implementation-planning":
|
|
6297
|
-
_validate_plan_body_state_file(
|
|
6298
|
-
_load_report_data(report_path) or {}, report_path, failures
|
|
6299
|
-
)
|
|
6676
|
+
_validate_plan_body_state_file(validation_data, report_path, failures)
|
|
6300
6677
|
if task_type == "final-verification":
|
|
6301
6678
|
_validate_verification_target_match(
|
|
6302
|
-
|
|
6679
|
+
validation_data,
|
|
6303
6680
|
run_manifest,
|
|
6304
6681
|
project_root,
|
|
6305
6682
|
failures,
|
|
6306
6683
|
)
|
|
6307
6684
|
# Phase-agnostic: any task-type can be launched with a carry-in.
|
|
6308
6685
|
_validate_clarification_carry_in_recorded(
|
|
6309
|
-
|
|
6686
|
+
validation_data,
|
|
6310
6687
|
report_path,
|
|
6311
6688
|
run_manifest_path,
|
|
6312
6689
|
project_root,
|
|
@@ -6315,7 +6692,12 @@ def main() -> int:
|
|
|
6315
6692
|
|
|
6316
6693
|
validation_status = "passed" if not failures else "failed"
|
|
6317
6694
|
update_validation_metadata(
|
|
6318
|
-
team_state,
|
|
6695
|
+
team_state,
|
|
6696
|
+
run_manifest,
|
|
6697
|
+
task_manifest,
|
|
6698
|
+
validation_status,
|
|
6699
|
+
failures,
|
|
6700
|
+
report_data=validation_data,
|
|
6319
6701
|
)
|
|
6320
6702
|
|
|
6321
6703
|
write_json(team_state_path, team_state)
|