okstra 0.170.3 → 0.172.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +13 -0
- package/docs/cli.md +4 -2
- package/docs/for-ai/skills/okstra-user-response.md +2 -2
- package/docs/project-structure-overview.md +3 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/launch.template.md +4 -0
- package/runtime/prompts/lead/adapters/cmux.md +1 -1
- package/runtime/prompts/lead/okstra-lead-contract.md +36 -12
- package/runtime/prompts/lead/plan-body-verification.md +22 -11
- package/runtime/prompts/lead/report-writer.md +11 -10
- package/runtime/prompts/lead/team-contract.md +2 -0
- package/runtime/prompts/profiles/_clarification-recommendation.md +3 -1
- package/runtime/prompts/profiles/_common-contract.md +2 -1
- package/runtime/prompts/profiles/implementation-planning.md +8 -1
- package/runtime/python/okstra_ctl/adapters/hosts/antigravity/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/kimi/relay.md +1 -1
- package/runtime/python/okstra_ctl/agent_activity.py +306 -0
- package/runtime/python/okstra_ctl/clarification_items.py +37 -20
- package/runtime/python/okstra_ctl/cmux.py +144 -59
- package/runtime/python/okstra_ctl/lead_events.py +47 -4
- package/runtime/python/okstra_ctl/render.py +11 -3
- package/runtime/python/okstra_ctl/report_finalize.py +51 -14
- package/runtime/python/okstra_ctl/report_html/common.py +5 -3
- package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +17 -1
- package/runtime/python/okstra_ctl/report_translation.py +14 -0
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +150 -0
- package/runtime/schemas/final-report-v2.0.schema.json +189 -0
- package/runtime/skills/okstra-user-response/SKILL.md +2 -2
- package/runtime/templates/reports/final-report-v2.template.md +8 -0
- package/runtime/templates/reports/html/assets/base.css +7 -0
- package/runtime/templates/reports/html/i18n/en.json +6 -1
- package/runtime/templates/reports/html/i18n/ko.json +6 -1
- package/runtime/templates/reports/html/macros/forms.html +21 -2
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +25 -0
- package/runtime/templates/reports/i18n/en.json +4 -0
- package/runtime/templates/reports/report.js +26 -17
- package/runtime/templates/reports/user-response.template.md +3 -1
- package/runtime/templates/worker-prompt-preamble.md +8 -0
- package/runtime/validators/validate-run.py +989 -29
- package/runtime/validators/validate_session_conformance.py +523 -35
- package/src/cli-registry.mjs +7 -0
- package/src/commands/report/agent-activity.mjs +21 -0
|
@@ -113,6 +113,7 @@ from okstra_ctl.agent_invocation import ( # noqa: E402
|
|
|
113
113
|
agent_model_assignment_from_payload,
|
|
114
114
|
verify_agent_invocation,
|
|
115
115
|
)
|
|
116
|
+
from okstra_ctl.lead_events import LeadEventParseError, read_lead_events # noqa: E402
|
|
116
117
|
from okstra_ctl.worker_audit_ledger import ( # noqa: E402
|
|
117
118
|
READING_CONFIRMATION_HEADING_RE,
|
|
118
119
|
check_worker_results_audit,
|
|
@@ -3343,11 +3344,15 @@ def validate_final_report_data(
|
|
|
3343
3344
|
if errors:
|
|
3344
3345
|
return data
|
|
3345
3346
|
|
|
3347
|
+
manifest = run_manifest or {}
|
|
3348
|
+
_validate_approval_context(data, manifest, failures, report_path)
|
|
3349
|
+
_validate_activity_contract_plan_limits(data, manifest, failures)
|
|
3350
|
+
|
|
3346
3351
|
analysis_result = validate_analysis_report(
|
|
3347
3352
|
data=data,
|
|
3348
3353
|
report_path=report_path,
|
|
3349
3354
|
project_root=project_root or report_path.parent,
|
|
3350
|
-
run_manifest=
|
|
3355
|
+
run_manifest=manifest,
|
|
3351
3356
|
clarification_text=clarification_text,
|
|
3352
3357
|
)
|
|
3353
3358
|
manifest_task_type = str((run_manifest or {}).get("taskType") or "")
|
|
@@ -3750,32 +3755,66 @@ def _state_classification(item: dict, gate_class: str) -> str:
|
|
|
3750
3755
|
return "dissent-isolated" if dissenting == 1 else "partial-consensus"
|
|
3751
3756
|
|
|
3752
3757
|
|
|
3753
|
-
def
|
|
3758
|
+
def _resolved_noncritical_dissent_ids(data: dict) -> set[str]:
|
|
3759
|
+
"""Plan items whose remaining dissent the user explicitly accepted."""
|
|
3760
|
+
accepted: set[str] = set()
|
|
3761
|
+
for row in data.get("clarificationItems") or []:
|
|
3762
|
+
if not isinstance(row, dict) or row.get("blocks") != "approval":
|
|
3763
|
+
continue
|
|
3764
|
+
context = row.get("approvalContext")
|
|
3765
|
+
if not isinstance(context, dict):
|
|
3766
|
+
continue
|
|
3767
|
+
resolution = context.get("resolution")
|
|
3768
|
+
if (
|
|
3769
|
+
row.get("status") == "resolved"
|
|
3770
|
+
and context.get("classification") == "noncritical-dissent"
|
|
3771
|
+
and isinstance(resolution, dict)
|
|
3772
|
+
and resolution.get("disposition") == "accept-risk"
|
|
3773
|
+
and str(resolution.get("userText") or "").strip()
|
|
3774
|
+
and _approval_context_activity_refs_exist(
|
|
3775
|
+
data, str(row.get("id") or ""), context, resolution
|
|
3776
|
+
)
|
|
3777
|
+
):
|
|
3778
|
+
accepted.update(
|
|
3779
|
+
item_id
|
|
3780
|
+
for item_id in context.get("planItemIds") or []
|
|
3781
|
+
if isinstance(item_id, str)
|
|
3782
|
+
)
|
|
3783
|
+
return accepted
|
|
3784
|
+
|
|
3785
|
+
|
|
3786
|
+
def _is_dissent_downgraded(
|
|
3787
|
+
item: dict,
|
|
3788
|
+
pbv: dict,
|
|
3789
|
+
accepted_item_ids: set[str],
|
|
3790
|
+
) -> bool:
|
|
3754
3791
|
"""Whether a surviving `majority-disagree` item stops blocking approval.
|
|
3755
3792
|
|
|
3756
|
-
|
|
3757
|
-
|
|
3758
|
-
|
|
3759
|
-
|
|
3760
|
-
instead of blocking the gate. Defects that would make the implementation
|
|
3761
|
-
itself wrong or unsafe (`_is_correctness_critical`) are excluded and keep
|
|
3762
|
-
blocking, so correctness never trades away for throughput.
|
|
3793
|
+
Exhausting the automatic self-fix budget records the unresolved dissent but
|
|
3794
|
+
does not accept it. Only an explicit, resolved noncritical risk-acceptance
|
|
3795
|
+
row can lower the item to `has-dissent`. Correctness-critical defects remain
|
|
3796
|
+
blocking regardless of the user's selected disposition.
|
|
3763
3797
|
"""
|
|
3764
3798
|
return (
|
|
3765
3799
|
_classify_plan_item_gate(item) == "majority-disagree"
|
|
3766
3800
|
and not _is_correctness_critical(item)
|
|
3767
3801
|
and _has_planner_fixable_majority(item)
|
|
3768
3802
|
and _self_fix_budget_exhausted(pbv)
|
|
3803
|
+
and str(item.get("id") or "") in accepted_item_ids
|
|
3769
3804
|
)
|
|
3770
3805
|
|
|
3771
3806
|
|
|
3772
|
-
def _recompute_plan_body_gate(
|
|
3807
|
+
def _recompute_plan_body_gate(
|
|
3808
|
+
pbv: dict,
|
|
3809
|
+
accepted_item_ids: set[str] | None = None,
|
|
3810
|
+
) -> str | None:
|
|
3773
3811
|
"""Recompute the whole §5.5.9 gate value from ``planItems[].verdicts``.
|
|
3774
3812
|
Returns a value in ``PLAN_VERIFY_GATE_VALUES`` or ``None`` when there are
|
|
3775
3813
|
no plan items to judge (disabled / empty round)."""
|
|
3814
|
+
accepted = accepted_item_ids or set()
|
|
3776
3815
|
classes = [
|
|
3777
3816
|
"has-dissent"
|
|
3778
|
-
if _is_dissent_downgraded(it, pbv)
|
|
3817
|
+
if _is_dissent_downgraded(it, pbv, accepted)
|
|
3779
3818
|
else _classify_plan_item_gate(it)
|
|
3780
3819
|
for it in (pbv.get("planItems") or [])
|
|
3781
3820
|
if isinstance(it, dict)
|
|
@@ -3791,7 +3830,11 @@ def _recompute_plan_body_gate(pbv: dict) -> str | None:
|
|
|
3791
3830
|
return "passed"
|
|
3792
3831
|
|
|
3793
3832
|
|
|
3794
|
-
def _validate_plan_body_gate_recompute(
|
|
3833
|
+
def _validate_plan_body_gate_recompute(
|
|
3834
|
+
data: dict,
|
|
3835
|
+
failures: list[str],
|
|
3836
|
+
accepted_item_ids: set[str] | None = None,
|
|
3837
|
+
) -> None:
|
|
3795
3838
|
"""H1 — the declared `Gate result` must not claim a healthier outcome than
|
|
3796
3839
|
the recorded per-worker verdicts support. Closes the forgery hole where a
|
|
3797
3840
|
lead writes `gateResult: passed` while workers actually voted DISAGREE:
|
|
@@ -3805,7 +3848,12 @@ def _validate_plan_body_gate_recompute(data: dict, failures: list[str]) -> None:
|
|
|
3805
3848
|
if not isinstance(pbv, dict):
|
|
3806
3849
|
return
|
|
3807
3850
|
declared = str(pbv.get("gateResult") or "").strip().lower()
|
|
3808
|
-
|
|
3851
|
+
accepted = (
|
|
3852
|
+
_resolved_noncritical_dissent_ids(data)
|
|
3853
|
+
if accepted_item_ids is None
|
|
3854
|
+
else accepted_item_ids
|
|
3855
|
+
)
|
|
3856
|
+
recomputed = _recompute_plan_body_gate(pbv, accepted)
|
|
3809
3857
|
if recomputed is None or declared not in _PLAN_GATE_RANK:
|
|
3810
3858
|
return
|
|
3811
3859
|
if _PLAN_GATE_RANK[declared] > _PLAN_GATE_RANK[recomputed]:
|
|
@@ -3869,10 +3917,14 @@ def _independent_coverage_blockers(ip: dict, pbv: dict) -> list[str]:
|
|
|
3869
3917
|
]
|
|
3870
3918
|
|
|
3871
3919
|
|
|
3872
|
-
def _gate_blocking_causes(
|
|
3920
|
+
def _gate_blocking_causes(
|
|
3921
|
+
pbv: dict,
|
|
3922
|
+
coverage_blockers: list[str],
|
|
3923
|
+
accepted_item_ids: set[str] | None = None,
|
|
3924
|
+
) -> set[str]:
|
|
3873
3925
|
"""Which inputs actually block approval, as `gateBlockedBy` enum values."""
|
|
3874
3926
|
causes = set()
|
|
3875
|
-
recomputed = _recompute_plan_body_gate(pbv)
|
|
3927
|
+
recomputed = _recompute_plan_body_gate(pbv, accepted_item_ids)
|
|
3876
3928
|
if recomputed == "blocked-by-disagreement":
|
|
3877
3929
|
causes.add("majority-disagree")
|
|
3878
3930
|
elif recomputed == "aborted-non-result":
|
|
@@ -3882,6 +3934,885 @@ def _gate_blocking_causes(pbv: dict, coverage_blockers: list[str]) -> set[str]:
|
|
|
3882
3934
|
return causes
|
|
3883
3935
|
|
|
3884
3936
|
|
|
3937
|
+
_APPROVAL_DISPOSITIONS_BY_CLASSIFICATION = {
|
|
3938
|
+
"user-decision": frozenset({"select", "request-revision", "reject"}),
|
|
3939
|
+
"noncritical-dissent": frozenset(
|
|
3940
|
+
{"accept-risk", "request-revision", "reject"}
|
|
3941
|
+
),
|
|
3942
|
+
"correctness-critical": frozenset({"request-revision", "reject"}),
|
|
3943
|
+
}
|
|
3944
|
+
|
|
3945
|
+
|
|
3946
|
+
def _is_activity_contract_v1_planning(run_manifest: dict) -> bool:
|
|
3947
|
+
return (
|
|
3948
|
+
run_manifest.get("activityContractVersion") == 1
|
|
3949
|
+
and run_manifest.get("taskType") == "implementation-planning"
|
|
3950
|
+
)
|
|
3951
|
+
|
|
3952
|
+
|
|
3953
|
+
def _independent_coverage_clarification_ids(ip: dict, pbv: dict) -> set[str]:
|
|
3954
|
+
promoted = _plan_body_promoted_clarification_ids(pbv)
|
|
3955
|
+
return {
|
|
3956
|
+
clarification_id
|
|
3957
|
+
for row in (ip.get("requirementCoverage") or [])
|
|
3958
|
+
if isinstance(row, dict) and _blocks_approval(row)
|
|
3959
|
+
for clarification_id in [_cited_clarification_id(row)]
|
|
3960
|
+
if clarification_id and clarification_id not in promoted
|
|
3961
|
+
}
|
|
3962
|
+
|
|
3963
|
+
|
|
3964
|
+
def _expected_approval_classification(
|
|
3965
|
+
row: dict,
|
|
3966
|
+
plan_items_by_id: dict[str, dict],
|
|
3967
|
+
independent_coverage_clarification_ids: set[str],
|
|
3968
|
+
) -> str:
|
|
3969
|
+
linked = [
|
|
3970
|
+
plan_items_by_id[item_id]
|
|
3971
|
+
for item_id in (row.get("approvalContext") or {}).get("planItemIds") or []
|
|
3972
|
+
if item_id in plan_items_by_id
|
|
3973
|
+
]
|
|
3974
|
+
if any(_is_correctness_critical(item) for item in linked):
|
|
3975
|
+
return "correctness-critical"
|
|
3976
|
+
if str(row.get("id") or "") in independent_coverage_clarification_ids:
|
|
3977
|
+
return "correctness-critical"
|
|
3978
|
+
for item in linked:
|
|
3979
|
+
disagree_votes = [
|
|
3980
|
+
verdict
|
|
3981
|
+
for verdict in (item.get("verdicts") or [])
|
|
3982
|
+
if isinstance(verdict, dict)
|
|
3983
|
+
and str(verdict.get("verdict") or "").upper() == "DISAGREE"
|
|
3984
|
+
]
|
|
3985
|
+
needs_user_input = sum(
|
|
3986
|
+
verdict.get("fixability") == "needs-user-input"
|
|
3987
|
+
for verdict in disagree_votes
|
|
3988
|
+
)
|
|
3989
|
+
if disagree_votes and needs_user_input * 2 > len(disagree_votes):
|
|
3990
|
+
return "user-decision"
|
|
3991
|
+
if any(_classify_plan_item_gate(item) == "majority-disagree" for item in linked):
|
|
3992
|
+
return "noncritical-dissent"
|
|
3993
|
+
return "user-decision"
|
|
3994
|
+
|
|
3995
|
+
|
|
3996
|
+
_STATE_DISAGREE_VOTE_RE = re.compile(r"^DISAGREE\(([a-f])\)$")
|
|
3997
|
+
_APPROVAL_CLARIFICATION_ID_RE = re.compile(r"^C-\d{3,}$")
|
|
3998
|
+
|
|
3999
|
+
|
|
4000
|
+
def _state_round_as_plan_item(item_id: str, round_row: dict) -> dict:
|
|
4001
|
+
verdicts = []
|
|
4002
|
+
votes = round_row.get("votes")
|
|
4003
|
+
for worker, raw_vote in (votes.items() if isinstance(votes, dict) else ()):
|
|
4004
|
+
vote = str(raw_vote or "").strip()
|
|
4005
|
+
match = _STATE_DISAGREE_VOTE_RE.fullmatch(vote)
|
|
4006
|
+
if match:
|
|
4007
|
+
verdicts.append(
|
|
4008
|
+
{"worker": worker, "verdict": "DISAGREE", "breakageKind": match.group(1)}
|
|
4009
|
+
)
|
|
4010
|
+
elif vote in {"AGREE", "SUPPLEMENT", "verification-error"}:
|
|
4011
|
+
verdicts.append({"worker": worker, "verdict": vote})
|
|
4012
|
+
return {"id": item_id, "verdicts": verdicts}
|
|
4013
|
+
|
|
4014
|
+
|
|
4015
|
+
def _historical_plan_item_evidence(state: dict) -> tuple[dict[str, str], set[str]]:
|
|
4016
|
+
classifications: dict[str, str] = {}
|
|
4017
|
+
item_ids: set[str] = set()
|
|
4018
|
+
for item in state.get("planItems") or []:
|
|
4019
|
+
if not isinstance(item, dict):
|
|
4020
|
+
continue
|
|
4021
|
+
item_id = str(item.get("id") or "").strip()
|
|
4022
|
+
if not item_id:
|
|
4023
|
+
continue
|
|
4024
|
+
item_ids.add(item_id)
|
|
4025
|
+
for round_row in item.get("rounds") or []:
|
|
4026
|
+
if not isinstance(round_row, dict):
|
|
4027
|
+
continue
|
|
4028
|
+
historical = _state_round_as_plan_item(item_id, round_row)
|
|
4029
|
+
if _is_correctness_critical(historical):
|
|
4030
|
+
classifications[item_id] = "correctness-critical"
|
|
4031
|
+
break
|
|
4032
|
+
if _classify_plan_item_gate(historical) == "majority-disagree":
|
|
4033
|
+
classifications.setdefault(item_id, "noncritical-dissent")
|
|
4034
|
+
return classifications, item_ids
|
|
4035
|
+
|
|
4036
|
+
|
|
4037
|
+
def _historical_coverage_clarification_ids(
|
|
4038
|
+
state: dict,
|
|
4039
|
+
plan_classifications: dict[str, str],
|
|
4040
|
+
) -> set[str]:
|
|
4041
|
+
coverage_gap_rounds = {
|
|
4042
|
+
round_row.get("round")
|
|
4043
|
+
for round_row in (state.get("roundHistory") or [])
|
|
4044
|
+
if isinstance(round_row, dict)
|
|
4045
|
+
and isinstance(round_row.get("round"), int)
|
|
4046
|
+
and "coverage-gap" in (round_row.get("gateBlockedBy") or [])
|
|
4047
|
+
}
|
|
4048
|
+
return {
|
|
4049
|
+
clarification_id
|
|
4050
|
+
for item in (state.get("planItems") or [])
|
|
4051
|
+
if isinstance(item, dict)
|
|
4052
|
+
for item_id in [str(item.get("id") or "").strip()]
|
|
4053
|
+
for clarification_id in [str(item.get("clarificationId") or "").strip()]
|
|
4054
|
+
if item_id
|
|
4055
|
+
and item_id not in plan_classifications
|
|
4056
|
+
and _APPROVAL_CLARIFICATION_ID_RE.fullmatch(clarification_id)
|
|
4057
|
+
and any(
|
|
4058
|
+
isinstance(round_row, dict)
|
|
4059
|
+
and round_row.get("round") in coverage_gap_rounds
|
|
4060
|
+
for round_row in (item.get("rounds") or [])
|
|
4061
|
+
)
|
|
4062
|
+
}
|
|
4063
|
+
|
|
4064
|
+
|
|
4065
|
+
def _read_approval_history(
|
|
4066
|
+
report_path: Path | None,
|
|
4067
|
+
) -> tuple[dict[str, str], set[str], set[str], dict, Path | None]:
|
|
4068
|
+
if report_path is None or (seq := _report_run_seq(report_path)) is None:
|
|
4069
|
+
return {}, set(), set(), {}, None
|
|
4070
|
+
state_path = (
|
|
4071
|
+
report_path.parent.parent
|
|
4072
|
+
/ "state"
|
|
4073
|
+
/ f"plan-body-verification-implementation-planning-{seq}.json"
|
|
4074
|
+
)
|
|
4075
|
+
try:
|
|
4076
|
+
state = json.loads(state_path.read_text(encoding="utf-8"))
|
|
4077
|
+
except (OSError, json.JSONDecodeError):
|
|
4078
|
+
return {}, set(), set(), {}, None
|
|
4079
|
+
if not isinstance(state, dict):
|
|
4080
|
+
return {}, set(), set(), {}, None
|
|
4081
|
+
classifications, item_ids = _historical_plan_item_evidence(state)
|
|
4082
|
+
coverage_ids = _historical_coverage_clarification_ids(
|
|
4083
|
+
state,
|
|
4084
|
+
classifications,
|
|
4085
|
+
)
|
|
4086
|
+
return classifications, item_ids, coverage_ids, state, state_path
|
|
4087
|
+
|
|
4088
|
+
|
|
4089
|
+
def _nonblocking_coverage_clarification_ids(ip: dict) -> set[str]:
|
|
4090
|
+
return {
|
|
4091
|
+
ref
|
|
4092
|
+
for row in (ip.get("requirementCoverage") or [])
|
|
4093
|
+
if isinstance(row, dict) and not _blocks_approval(row)
|
|
4094
|
+
for ref in (row.get("decisionRefs") or [])
|
|
4095
|
+
if isinstance(ref, str) and _APPROVAL_CLARIFICATION_ID_RE.fullmatch(ref)
|
|
4096
|
+
}
|
|
4097
|
+
|
|
4098
|
+
|
|
4099
|
+
def _historical_approval_classification(
|
|
4100
|
+
row_id: str,
|
|
4101
|
+
linked_ids: list[str],
|
|
4102
|
+
historical_plan_classifications: dict[str, str],
|
|
4103
|
+
historical_coverage_ids: set[str],
|
|
4104
|
+
) -> str | None:
|
|
4105
|
+
if row_id in historical_coverage_ids:
|
|
4106
|
+
return "correctness-critical"
|
|
4107
|
+
classes = {
|
|
4108
|
+
historical_plan_classifications[item_id]
|
|
4109
|
+
for item_id in linked_ids
|
|
4110
|
+
if item_id in historical_plan_classifications
|
|
4111
|
+
}
|
|
4112
|
+
if "correctness-critical" in classes:
|
|
4113
|
+
return "correctness-critical"
|
|
4114
|
+
if "noncritical-dissent" in classes:
|
|
4115
|
+
return "noncritical-dissent"
|
|
4116
|
+
return None
|
|
4117
|
+
|
|
4118
|
+
|
|
4119
|
+
def _approval_activities_by_id(data: dict) -> dict[str, dict]:
|
|
4120
|
+
return {
|
|
4121
|
+
activity_id: activity
|
|
4122
|
+
for activity in (data.get("agentActivity") or [])
|
|
4123
|
+
if isinstance(activity, dict)
|
|
4124
|
+
for activity_id in [activity.get("activityId")]
|
|
4125
|
+
if isinstance(activity_id, str) and activity_id
|
|
4126
|
+
}
|
|
4127
|
+
|
|
4128
|
+
|
|
4129
|
+
def _canonical_activity_timestamps(
|
|
4130
|
+
run_manifest: Mapping[str, Any],
|
|
4131
|
+
report_path: Path | None,
|
|
4132
|
+
) -> dict[str, str]:
|
|
4133
|
+
raw_path = run_manifest.get("leadEventsPath")
|
|
4134
|
+
if not isinstance(raw_path, str) or not raw_path.strip():
|
|
4135
|
+
return {}
|
|
4136
|
+
path = Path(raw_path)
|
|
4137
|
+
if not path.is_absolute() and report_path is not None:
|
|
4138
|
+
path = _project_root_from_report(report_path) / path
|
|
4139
|
+
try:
|
|
4140
|
+
events = read_lead_events(path)
|
|
4141
|
+
except (LeadEventParseError, OSError):
|
|
4142
|
+
return {}
|
|
4143
|
+
return {
|
|
4144
|
+
str(event.details.get("activityId")): event.timestamp
|
|
4145
|
+
for event in events
|
|
4146
|
+
if event.event_type == "activity"
|
|
4147
|
+
and isinstance(event.details.get("activityId"), str)
|
|
4148
|
+
}
|
|
4149
|
+
|
|
4150
|
+
|
|
4151
|
+
def _is_decision_required_activity(activity: dict | None) -> bool:
|
|
4152
|
+
return bool(
|
|
4153
|
+
activity
|
|
4154
|
+
and activity.get("kind") == "user-decision-required"
|
|
4155
|
+
and activity.get("outcome") == "blocked"
|
|
4156
|
+
)
|
|
4157
|
+
|
|
4158
|
+
|
|
4159
|
+
def _is_applied_decision_check(activity: dict | None) -> bool:
|
|
4160
|
+
commands = (activity or {}).get("commands")
|
|
4161
|
+
return bool(
|
|
4162
|
+
activity
|
|
4163
|
+
and activity.get("kind") == "user-decision-evaluated"
|
|
4164
|
+
and activity.get("outcome") == "resolved"
|
|
4165
|
+
and str(activity.get("resultPath") or "").strip()
|
|
4166
|
+
and isinstance(commands, list)
|
|
4167
|
+
and bool(commands)
|
|
4168
|
+
and all(
|
|
4169
|
+
isinstance(command, dict) and command.get("exitCode") == 0
|
|
4170
|
+
for command in commands
|
|
4171
|
+
)
|
|
4172
|
+
)
|
|
4173
|
+
|
|
4174
|
+
|
|
4175
|
+
def _activity_matches_approval_context(
|
|
4176
|
+
activity: dict | None,
|
|
4177
|
+
row_id: str,
|
|
4178
|
+
context: dict,
|
|
4179
|
+
) -> bool:
|
|
4180
|
+
if not activity:
|
|
4181
|
+
return False
|
|
4182
|
+
evidence_refs = {
|
|
4183
|
+
ref
|
|
4184
|
+
for ref in (activity.get("evidenceRefs") or [])
|
|
4185
|
+
if isinstance(ref, str)
|
|
4186
|
+
}
|
|
4187
|
+
activity_item_ids = {
|
|
4188
|
+
item_id
|
|
4189
|
+
for item_id in (activity.get("planItemIds") or [])
|
|
4190
|
+
if isinstance(item_id, str)
|
|
4191
|
+
}
|
|
4192
|
+
context_item_ids = {
|
|
4193
|
+
item_id
|
|
4194
|
+
for item_id in (context.get("planItemIds") or [])
|
|
4195
|
+
if isinstance(item_id, str)
|
|
4196
|
+
}
|
|
4197
|
+
clarification_refs = {
|
|
4198
|
+
ref for ref in evidence_refs if _APPROVAL_CLARIFICATION_ID_RE.fullmatch(ref)
|
|
4199
|
+
}
|
|
4200
|
+
return clarification_refs == {row_id} and context_item_ids == activity_item_ids
|
|
4201
|
+
|
|
4202
|
+
|
|
4203
|
+
_TARGETED_REVERIFICATION_REF_RE = re.compile(
|
|
4204
|
+
r"^plan-body-verification:round-(?P<round>\d+)$"
|
|
4205
|
+
)
|
|
4206
|
+
|
|
4207
|
+
|
|
4208
|
+
def _targeted_reverification_round(activity: dict | None) -> int | None:
|
|
4209
|
+
rounds = {
|
|
4210
|
+
int(match.group("round"))
|
|
4211
|
+
for ref in ((activity or {}).get("evidenceRefs") or [])
|
|
4212
|
+
if isinstance(ref, str)
|
|
4213
|
+
for match in [_TARGETED_REVERIFICATION_REF_RE.fullmatch(ref)]
|
|
4214
|
+
if match is not None
|
|
4215
|
+
}
|
|
4216
|
+
if len(rounds) != 1:
|
|
4217
|
+
return None
|
|
4218
|
+
return next(iter(rounds))
|
|
4219
|
+
|
|
4220
|
+
|
|
4221
|
+
def _approval_context_activity_refs_exist(
|
|
4222
|
+
data: dict,
|
|
4223
|
+
row_id: str,
|
|
4224
|
+
context: dict,
|
|
4225
|
+
resolution: dict,
|
|
4226
|
+
) -> bool:
|
|
4227
|
+
activities = _approval_activities_by_id(data)
|
|
4228
|
+
activity_ids = {
|
|
4229
|
+
value for value in (context.get("activityIds") or []) if isinstance(value, str)
|
|
4230
|
+
}
|
|
4231
|
+
check_refs = {
|
|
4232
|
+
value for value in (resolution.get("checkRefs") or []) if isinstance(value, str)
|
|
4233
|
+
}
|
|
4234
|
+
activity_order = {
|
|
4235
|
+
activity.get("activityId"): index
|
|
4236
|
+
for index, activity in enumerate(data.get("agentActivity") or [])
|
|
4237
|
+
if isinstance(activity, dict)
|
|
4238
|
+
}
|
|
4239
|
+
ordered = bool(activity_ids and check_refs) and max(
|
|
4240
|
+
activity_order.get(ref, -1) for ref in activity_ids
|
|
4241
|
+
) < min(activity_order.get(ref, -1) for ref in check_refs)
|
|
4242
|
+
return (
|
|
4243
|
+
bool(activity_ids)
|
|
4244
|
+
and bool(check_refs)
|
|
4245
|
+
and all(
|
|
4246
|
+
_is_decision_required_activity(activities.get(ref))
|
|
4247
|
+
and _activity_matches_approval_context(
|
|
4248
|
+
activities.get(ref), row_id, context
|
|
4249
|
+
)
|
|
4250
|
+
for ref in activity_ids
|
|
4251
|
+
)
|
|
4252
|
+
and all(
|
|
4253
|
+
_is_applied_decision_check(activities.get(ref))
|
|
4254
|
+
and _targeted_reverification_round(activities.get(ref)) is not None
|
|
4255
|
+
and _activity_matches_approval_context(
|
|
4256
|
+
activities.get(ref), row_id, context
|
|
4257
|
+
)
|
|
4258
|
+
for ref in check_refs
|
|
4259
|
+
)
|
|
4260
|
+
and ordered
|
|
4261
|
+
)
|
|
4262
|
+
|
|
4263
|
+
|
|
4264
|
+
def _validate_approval_activity_refs(
|
|
4265
|
+
row_id: str,
|
|
4266
|
+
context: dict,
|
|
4267
|
+
activities: dict[str, dict],
|
|
4268
|
+
failures: list[str],
|
|
4269
|
+
) -> None:
|
|
4270
|
+
activity_ids = {
|
|
4271
|
+
value for value in (context.get("activityIds") or []) if isinstance(value, str)
|
|
4272
|
+
}
|
|
4273
|
+
unknown_activity_ids = sorted(activity_ids - set(activities))
|
|
4274
|
+
if not activity_ids or unknown_activity_ids:
|
|
4275
|
+
failures.append(
|
|
4276
|
+
f"final-report data.json: approval clarification `{row_id}` activityIds "
|
|
4277
|
+
f"must reference agentActivity[].activityId values; unknown="
|
|
4278
|
+
f"{unknown_activity_ids or 'none'}, recorded={sorted(activity_ids)}."
|
|
4279
|
+
)
|
|
4280
|
+
elif not all(
|
|
4281
|
+
_is_decision_required_activity(activities.get(ref)) for ref in activity_ids
|
|
4282
|
+
):
|
|
4283
|
+
failures.append(
|
|
4284
|
+
f"final-report data.json: approval clarification `{row_id}` activityIds "
|
|
4285
|
+
"must reference blocked user-decision-required activities."
|
|
4286
|
+
)
|
|
4287
|
+
elif not all(
|
|
4288
|
+
_activity_matches_approval_context(activities.get(ref), row_id, context)
|
|
4289
|
+
for ref in activity_ids
|
|
4290
|
+
):
|
|
4291
|
+
failures.append(
|
|
4292
|
+
f"final-report data.json: approval clarification `{row_id}` activityIds "
|
|
4293
|
+
f"must cite exactly one clarification (`{row_id}`) in evidenceRefs "
|
|
4294
|
+
"and exactly match approvalContext.planItemIds."
|
|
4295
|
+
)
|
|
4296
|
+
resolution = context.get("resolution")
|
|
4297
|
+
if not isinstance(resolution, dict):
|
|
4298
|
+
return
|
|
4299
|
+
check_refs = {
|
|
4300
|
+
value for value in (resolution.get("checkRefs") or []) if isinstance(value, str)
|
|
4301
|
+
}
|
|
4302
|
+
unknown_check_refs = sorted(check_refs - set(activities))
|
|
4303
|
+
if check_refs and unknown_check_refs:
|
|
4304
|
+
failures.append(
|
|
4305
|
+
f"final-report data.json: approval clarification `{row_id}` resolution."
|
|
4306
|
+
f"checkRefs must reference agentActivity[].activityId values; unknown="
|
|
4307
|
+
f"{unknown_check_refs}."
|
|
4308
|
+
)
|
|
4309
|
+
elif check_refs and not all(
|
|
4310
|
+
_is_applied_decision_check(activities.get(ref))
|
|
4311
|
+
and _targeted_reverification_round(activities.get(ref)) is not None
|
|
4312
|
+
for ref in check_refs
|
|
4313
|
+
):
|
|
4314
|
+
failures.append(
|
|
4315
|
+
f"final-report data.json: approval clarification `{row_id}` resolution."
|
|
4316
|
+
"checkRefs must reference resolved user-decision-evaluated activities "
|
|
4317
|
+
"with successful check evidence and one "
|
|
4318
|
+
"`plan-body-verification:round-N` evidenceRef."
|
|
4319
|
+
)
|
|
4320
|
+
elif check_refs and not all(
|
|
4321
|
+
_activity_matches_approval_context(activities.get(ref), row_id, context)
|
|
4322
|
+
for ref in check_refs
|
|
4323
|
+
):
|
|
4324
|
+
failures.append(
|
|
4325
|
+
f"final-report data.json: approval clarification `{row_id}` resolution."
|
|
4326
|
+
f"checkRefs must cite exactly one clarification (`{row_id}`) in "
|
|
4327
|
+
"evidenceRefs and exactly match approvalContext.planItemIds."
|
|
4328
|
+
)
|
|
4329
|
+
elif check_refs:
|
|
4330
|
+
order = {
|
|
4331
|
+
activity.get("activityId"): index
|
|
4332
|
+
for index, activity in enumerate(activities.values())
|
|
4333
|
+
}
|
|
4334
|
+
if activity_ids and max(order.get(ref, -1) for ref in activity_ids) >= min(
|
|
4335
|
+
order.get(ref, -1) for ref in check_refs
|
|
4336
|
+
):
|
|
4337
|
+
failures.append(
|
|
4338
|
+
f"final-report data.json: approval clarification `{row_id}` "
|
|
4339
|
+
"user-decision-evaluated activity must occur after every "
|
|
4340
|
+
"user-decision-required activity."
|
|
4341
|
+
)
|
|
4342
|
+
|
|
4343
|
+
|
|
4344
|
+
def _validate_approval_dispositions(
|
|
4345
|
+
row: dict,
|
|
4346
|
+
context: dict,
|
|
4347
|
+
failures: list[str],
|
|
4348
|
+
) -> None:
|
|
4349
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4350
|
+
classification = str(context.get("classification") or "")
|
|
4351
|
+
allowed = _APPROVAL_DISPOSITIONS_BY_CLASSIFICATION.get(classification, frozenset())
|
|
4352
|
+
candidates = [("recommendedDisposition", context.get("recommendedDisposition"))]
|
|
4353
|
+
candidates.extend(
|
|
4354
|
+
(f"options[{index}].disposition", option.get("disposition"))
|
|
4355
|
+
for index, option in enumerate(row.get("options") or [])
|
|
4356
|
+
if isinstance(option, dict)
|
|
4357
|
+
)
|
|
4358
|
+
resolution = context.get("resolution")
|
|
4359
|
+
if isinstance(resolution, dict):
|
|
4360
|
+
candidates.append(("resolution.disposition", resolution.get("disposition")))
|
|
4361
|
+
for field, disposition in candidates:
|
|
4362
|
+
if disposition not in allowed:
|
|
4363
|
+
failures.append(
|
|
4364
|
+
f"final-report data.json: approval clarification `{row_id}` "
|
|
4365
|
+
f"classification `{classification}` does not allow `{disposition}` "
|
|
4366
|
+
f"in {field}; allowed dispositions are {sorted(allowed)}."
|
|
4367
|
+
)
|
|
4368
|
+
|
|
4369
|
+
|
|
4370
|
+
def _validate_resolved_approval(
|
|
4371
|
+
row: dict,
|
|
4372
|
+
context: dict,
|
|
4373
|
+
failures: list[str],
|
|
4374
|
+
) -> None:
|
|
4375
|
+
if row.get("status") != "resolved":
|
|
4376
|
+
return
|
|
4377
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4378
|
+
resolution = context.get("resolution")
|
|
4379
|
+
if not isinstance(resolution, dict):
|
|
4380
|
+
failures.append(
|
|
4381
|
+
f"final-report data.json: resolved approval clarification `{row_id}` "
|
|
4382
|
+
"requires resolution.userText and non-empty resolution.checkRefs."
|
|
4383
|
+
)
|
|
4384
|
+
return
|
|
4385
|
+
if not str(resolution.get("userText") or "").strip():
|
|
4386
|
+
failures.append(
|
|
4387
|
+
f"final-report data.json: resolved approval clarification `{row_id}` "
|
|
4388
|
+
"requires non-empty resolution.userText."
|
|
4389
|
+
)
|
|
4390
|
+
check_refs = resolution.get("checkRefs")
|
|
4391
|
+
if not isinstance(check_refs, list) or not any(
|
|
4392
|
+
isinstance(value, str) and value for value in check_refs
|
|
4393
|
+
):
|
|
4394
|
+
failures.append(
|
|
4395
|
+
f"final-report data.json: resolved approval clarification `{row_id}` "
|
|
4396
|
+
"requires non-empty resolution.checkRefs."
|
|
4397
|
+
)
|
|
4398
|
+
if (
|
|
4399
|
+
context.get("classification") == "noncritical-dissent"
|
|
4400
|
+
and resolution.get("disposition") != "accept-risk"
|
|
4401
|
+
):
|
|
4402
|
+
failures.append(
|
|
4403
|
+
f"final-report data.json: resolved noncritical-dissent `{row_id}` "
|
|
4404
|
+
"requires an explicit accept-risk disposition."
|
|
4405
|
+
)
|
|
4406
|
+
|
|
4407
|
+
|
|
4408
|
+
def _has_successful_targeted_reverification(item: dict) -> bool:
|
|
4409
|
+
verdicts = [
|
|
4410
|
+
str(verdict.get("verdict") or "").strip().upper()
|
|
4411
|
+
for verdict in (item.get("verdicts") or [])
|
|
4412
|
+
if isinstance(verdict, dict)
|
|
4413
|
+
]
|
|
4414
|
+
return bool(verdicts) and all(
|
|
4415
|
+
verdict in {"AGREE", "SUPPLEMENT"} for verdict in verdicts
|
|
4416
|
+
)
|
|
4417
|
+
|
|
4418
|
+
|
|
4419
|
+
def _parse_approval_timestamp(value: Any) -> datetime | None:
|
|
4420
|
+
if not isinstance(value, str) or not value.strip():
|
|
4421
|
+
return None
|
|
4422
|
+
try:
|
|
4423
|
+
parsed = datetime.fromisoformat(value.strip().replace("Z", "+00:00"))
|
|
4424
|
+
except ValueError:
|
|
4425
|
+
return None
|
|
4426
|
+
if parsed.tzinfo is None or parsed.utcoffset() != timezone.utc.utcoffset(parsed):
|
|
4427
|
+
return None
|
|
4428
|
+
return parsed
|
|
4429
|
+
|
|
4430
|
+
|
|
4431
|
+
def _referenced_approval_timestamps(
|
|
4432
|
+
refs: Any,
|
|
4433
|
+
activity_timestamps: dict[str, str],
|
|
4434
|
+
) -> list[datetime | None]:
|
|
4435
|
+
return [
|
|
4436
|
+
_parse_approval_timestamp(activity_timestamps.get(ref))
|
|
4437
|
+
for ref in (refs or [])
|
|
4438
|
+
if isinstance(ref, str)
|
|
4439
|
+
]
|
|
4440
|
+
|
|
4441
|
+
|
|
4442
|
+
def _target_round_completed_at(
|
|
4443
|
+
approval_state: dict,
|
|
4444
|
+
target_round: int,
|
|
4445
|
+
) -> datetime | None:
|
|
4446
|
+
matching_rounds = [
|
|
4447
|
+
row
|
|
4448
|
+
for row in (approval_state.get("roundHistory") or [])
|
|
4449
|
+
if isinstance(row, dict) and row.get("round") == target_round
|
|
4450
|
+
]
|
|
4451
|
+
if len(matching_rounds) != 1:
|
|
4452
|
+
return None
|
|
4453
|
+
return _parse_approval_timestamp(matching_rounds[0].get("completedAt"))
|
|
4454
|
+
|
|
4455
|
+
|
|
4456
|
+
def _validate_target_round_causality(
|
|
4457
|
+
row_id: str,
|
|
4458
|
+
target_round: int,
|
|
4459
|
+
approval_state: dict,
|
|
4460
|
+
context: dict,
|
|
4461
|
+
resolution: dict,
|
|
4462
|
+
activity_timestamps: dict[str, str],
|
|
4463
|
+
failures: list[str],
|
|
4464
|
+
) -> None:
|
|
4465
|
+
completed_at = _target_round_completed_at(approval_state, target_round)
|
|
4466
|
+
required_at = _referenced_approval_timestamps(
|
|
4467
|
+
context.get("activityIds"),
|
|
4468
|
+
activity_timestamps,
|
|
4469
|
+
)
|
|
4470
|
+
evaluated_at = _referenced_approval_timestamps(
|
|
4471
|
+
resolution.get("checkRefs"),
|
|
4472
|
+
activity_timestamps,
|
|
4473
|
+
)
|
|
4474
|
+
if (
|
|
4475
|
+
completed_at is None
|
|
4476
|
+
or not required_at
|
|
4477
|
+
or not evaluated_at
|
|
4478
|
+
or None in required_at
|
|
4479
|
+
or None in evaluated_at
|
|
4480
|
+
):
|
|
4481
|
+
failures.append(
|
|
4482
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4483
|
+
f"state round {target_round} requires one UTC completedAt plus canonical "
|
|
4484
|
+
"timestamps for every required and evaluated activity."
|
|
4485
|
+
)
|
|
4486
|
+
return
|
|
4487
|
+
if completed_at <= max(required_at):
|
|
4488
|
+
failures.append(
|
|
4489
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4490
|
+
f"state round {target_round} completedAt must be after every referenced "
|
|
4491
|
+
"user-decision-required activity."
|
|
4492
|
+
)
|
|
4493
|
+
if completed_at > min(evaluated_at):
|
|
4494
|
+
failures.append(
|
|
4495
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4496
|
+
f"state round {target_round} completedAt must be no later than every "
|
|
4497
|
+
"referenced user-decision-evaluated activity."
|
|
4498
|
+
)
|
|
4499
|
+
|
|
4500
|
+
|
|
4501
|
+
def _required_activity_plan_item_ids(
|
|
4502
|
+
context: dict,
|
|
4503
|
+
activities: dict[str, dict],
|
|
4504
|
+
) -> set[str]:
|
|
4505
|
+
return {
|
|
4506
|
+
item_id
|
|
4507
|
+
for ref in (context.get("activityIds") or [])
|
|
4508
|
+
if isinstance(ref, str) and _is_decision_required_activity(activities.get(ref))
|
|
4509
|
+
for item_id in (activities[ref].get("planItemIds") or [])
|
|
4510
|
+
if isinstance(item_id, str)
|
|
4511
|
+
}
|
|
4512
|
+
|
|
4513
|
+
|
|
4514
|
+
def _terminal_unknown_plan_items_are_historical(
|
|
4515
|
+
row: dict,
|
|
4516
|
+
context: dict,
|
|
4517
|
+
unknown_ids: set[str],
|
|
4518
|
+
historical_item_ids: set[str],
|
|
4519
|
+
historical_coverage_ids: set[str],
|
|
4520
|
+
activities: dict[str, dict],
|
|
4521
|
+
) -> bool:
|
|
4522
|
+
status = row.get("status")
|
|
4523
|
+
if status not in {"resolved", "obsolete"}:
|
|
4524
|
+
return False
|
|
4525
|
+
if status == "resolved" and str(row.get("id") or "") not in historical_coverage_ids:
|
|
4526
|
+
return False
|
|
4527
|
+
required_item_ids = _required_activity_plan_item_ids(context, activities)
|
|
4528
|
+
return bool(unknown_ids) and unknown_ids <= historical_item_ids & required_item_ids
|
|
4529
|
+
|
|
4530
|
+
|
|
4531
|
+
def _validate_correctness_resolution(
|
|
4532
|
+
row: dict,
|
|
4533
|
+
linked_ids: list[str],
|
|
4534
|
+
linked_items: list[dict],
|
|
4535
|
+
independent_coverage_clarification_ids: set[str],
|
|
4536
|
+
activities: dict[str, dict],
|
|
4537
|
+
approval_state: dict,
|
|
4538
|
+
approval_state_path: Path | None,
|
|
4539
|
+
activity_timestamps: dict[str, str],
|
|
4540
|
+
failures: list[str],
|
|
4541
|
+
) -> None:
|
|
4542
|
+
context = row.get("approvalContext") or {}
|
|
4543
|
+
if (
|
|
4544
|
+
context.get("classification") != "correctness-critical"
|
|
4545
|
+
or row.get("status") != "resolved"
|
|
4546
|
+
):
|
|
4547
|
+
return
|
|
4548
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4549
|
+
resolution = context.get("resolution") or {}
|
|
4550
|
+
target_rounds = {
|
|
4551
|
+
round_number
|
|
4552
|
+
for ref in (resolution.get("checkRefs") or [])
|
|
4553
|
+
if isinstance(ref, str)
|
|
4554
|
+
for round_number in [_targeted_reverification_round(activities.get(ref))]
|
|
4555
|
+
if round_number is not None
|
|
4556
|
+
}
|
|
4557
|
+
if len(target_rounds) != 1:
|
|
4558
|
+
failures.append(
|
|
4559
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4560
|
+
"requires exactly one evidenced targeted reverification state round."
|
|
4561
|
+
)
|
|
4562
|
+
return
|
|
4563
|
+
target_round = next(iter(target_rounds))
|
|
4564
|
+
_validate_target_round_causality(
|
|
4565
|
+
row_id,
|
|
4566
|
+
target_round,
|
|
4567
|
+
approval_state,
|
|
4568
|
+
context,
|
|
4569
|
+
resolution,
|
|
4570
|
+
activity_timestamps,
|
|
4571
|
+
failures,
|
|
4572
|
+
)
|
|
4573
|
+
state_name = approval_state_path.name if approval_state_path is not None else ""
|
|
4574
|
+
result_paths_match = bool(state_name) and all(
|
|
4575
|
+
tuple(
|
|
4576
|
+
Path(str(activities[ref].get("resultPath") or "")).parts[-2:]
|
|
4577
|
+
) == ("state", state_name)
|
|
4578
|
+
for ref in (resolution.get("checkRefs") or [])
|
|
4579
|
+
if isinstance(ref, str) and ref in activities
|
|
4580
|
+
)
|
|
4581
|
+
if not result_paths_match:
|
|
4582
|
+
failures.append(
|
|
4583
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4584
|
+
"evaluation resultPath must reference the matching plan-body "
|
|
4585
|
+
"verification state artifact."
|
|
4586
|
+
)
|
|
4587
|
+
state_items = {
|
|
4588
|
+
str(item.get("id") or ""): item
|
|
4589
|
+
for item in (approval_state.get("planItems") or [])
|
|
4590
|
+
if isinstance(item, dict) and str(item.get("id") or "")
|
|
4591
|
+
}
|
|
4592
|
+
state_failures: dict[str, str] = {}
|
|
4593
|
+
current_by_id = {str(item.get("id") or ""): item for item in linked_items}
|
|
4594
|
+
for item_id in linked_ids:
|
|
4595
|
+
state_item = state_items.get(item_id)
|
|
4596
|
+
rounds = state_item.get("rounds") if isinstance(state_item, dict) else None
|
|
4597
|
+
target_state_round = next(
|
|
4598
|
+
(
|
|
4599
|
+
round_row
|
|
4600
|
+
for round_row in (rounds or [])
|
|
4601
|
+
if isinstance(round_row, dict)
|
|
4602
|
+
and round_row.get("round") == target_round
|
|
4603
|
+
),
|
|
4604
|
+
None,
|
|
4605
|
+
)
|
|
4606
|
+
blocking_rounds = [
|
|
4607
|
+
round_row.get("round")
|
|
4608
|
+
for round_row in (rounds or [])
|
|
4609
|
+
if isinstance(round_row, dict)
|
|
4610
|
+
and isinstance(round_row.get("round"), int)
|
|
4611
|
+
and round_row.get("round") < target_round
|
|
4612
|
+
and (
|
|
4613
|
+
_is_correctness_critical(
|
|
4614
|
+
_state_round_as_plan_item(item_id, round_row)
|
|
4615
|
+
)
|
|
4616
|
+
or _classify_plan_item_gate(
|
|
4617
|
+
_state_round_as_plan_item(item_id, round_row)
|
|
4618
|
+
)
|
|
4619
|
+
== "majority-disagree"
|
|
4620
|
+
)
|
|
4621
|
+
]
|
|
4622
|
+
if (
|
|
4623
|
+
isinstance(state_item, dict)
|
|
4624
|
+
and state_item.get("clarificationId") == row_id
|
|
4625
|
+
):
|
|
4626
|
+
blocking_rounds.extend(
|
|
4627
|
+
round_row.get("round")
|
|
4628
|
+
for round_row in (approval_state.get("roundHistory") or [])
|
|
4629
|
+
if isinstance(round_row, dict)
|
|
4630
|
+
and isinstance(round_row.get("round"), int)
|
|
4631
|
+
and round_row.get("round") < target_round
|
|
4632
|
+
and "coverage-gap" in (round_row.get("gateBlockedBy") or [])
|
|
4633
|
+
)
|
|
4634
|
+
target_votes = (
|
|
4635
|
+
target_state_round.get("votes")
|
|
4636
|
+
if isinstance(target_state_round, dict)
|
|
4637
|
+
else None
|
|
4638
|
+
)
|
|
4639
|
+
successful = bool(target_votes) and all(
|
|
4640
|
+
vote in {"AGREE", "SUPPLEMENT"} for vote in target_votes.values()
|
|
4641
|
+
)
|
|
4642
|
+
if not blocking_rounds or not successful:
|
|
4643
|
+
state_failures[item_id] = (
|
|
4644
|
+
f"state round {target_round} is not a successful post-blocker round"
|
|
4645
|
+
)
|
|
4646
|
+
continue
|
|
4647
|
+
current = current_by_id.get(item_id)
|
|
4648
|
+
if current is not None:
|
|
4649
|
+
report_votes = {
|
|
4650
|
+
str(verdict.get("worker") or ""): str(verdict.get("verdict") or "")
|
|
4651
|
+
for verdict in (current.get("verdicts") or [])
|
|
4652
|
+
if isinstance(verdict, dict)
|
|
4653
|
+
}
|
|
4654
|
+
if report_votes != target_votes:
|
|
4655
|
+
state_failures[item_id] = (
|
|
4656
|
+
f"state round {target_round} votes do not match final report verdicts"
|
|
4657
|
+
)
|
|
4658
|
+
if state_failures:
|
|
4659
|
+
failures.append(
|
|
4660
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4661
|
+
f"targeted reverification state round {target_round} is not bound to "
|
|
4662
|
+
f"the resolved evidence; failures={state_failures}."
|
|
4663
|
+
)
|
|
4664
|
+
unresolved = {
|
|
4665
|
+
str(item.get("id") or "<unknown>"): [
|
|
4666
|
+
str(verdict.get("verdict") or "")
|
|
4667
|
+
for verdict in (item.get("verdicts") or [])
|
|
4668
|
+
if isinstance(verdict, dict)
|
|
4669
|
+
]
|
|
4670
|
+
for item in linked_items
|
|
4671
|
+
if not _has_successful_targeted_reverification(item)
|
|
4672
|
+
}
|
|
4673
|
+
if unresolved:
|
|
4674
|
+
failures.append(
|
|
4675
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4676
|
+
"cannot resolve until targeted reverification records only AGREE or "
|
|
4677
|
+
f"acceptable SUPPLEMENT verdicts; unresolved plan items={unresolved}."
|
|
4678
|
+
)
|
|
4679
|
+
if row_id in independent_coverage_clarification_ids:
|
|
4680
|
+
failures.append(
|
|
4681
|
+
f"final-report data.json: correctness-critical clarification `{row_id}` "
|
|
4682
|
+
"cannot resolve while an independent requirement coverage blocker remains."
|
|
4683
|
+
)
|
|
4684
|
+
|
|
4685
|
+
|
|
4686
|
+
def _validate_approval_context(
|
|
4687
|
+
data: dict,
|
|
4688
|
+
run_manifest: dict,
|
|
4689
|
+
failures: list[str],
|
|
4690
|
+
report_path: Path | None = None,
|
|
4691
|
+
) -> None:
|
|
4692
|
+
if not _is_activity_contract_v1_planning(run_manifest):
|
|
4693
|
+
return
|
|
4694
|
+
ip = data.get("implementationPlanning") or {}
|
|
4695
|
+
pbv = ip.get("planBodyVerification") or {}
|
|
4696
|
+
plan_items_by_id = {
|
|
4697
|
+
str(item.get("id")): item
|
|
4698
|
+
for item in (pbv.get("planItems") or [])
|
|
4699
|
+
if isinstance(item, dict) and str(item.get("id") or "")
|
|
4700
|
+
}
|
|
4701
|
+
coverage_ids = _independent_coverage_clarification_ids(ip, pbv)
|
|
4702
|
+
(
|
|
4703
|
+
historical_classes,
|
|
4704
|
+
historical_item_ids,
|
|
4705
|
+
historical_coverage_ids,
|
|
4706
|
+
approval_state,
|
|
4707
|
+
approval_state_path,
|
|
4708
|
+
) = _read_approval_history(report_path)
|
|
4709
|
+
historical_coverage_ids &= _nonblocking_coverage_clarification_ids(
|
|
4710
|
+
ip
|
|
4711
|
+
)
|
|
4712
|
+
activities = _approval_activities_by_id(data)
|
|
4713
|
+
activity_timestamps = _canonical_activity_timestamps(run_manifest, report_path)
|
|
4714
|
+
report_approved = (data.get("frontmatter") or {}).get("approved") is True
|
|
4715
|
+
for row in data.get("clarificationItems") or []:
|
|
4716
|
+
if not isinstance(row, dict) or row.get("blocks") != "approval":
|
|
4717
|
+
continue
|
|
4718
|
+
row_id = str(row.get("id") or "<unknown>")
|
|
4719
|
+
context = row.get("approvalContext")
|
|
4720
|
+
if not isinstance(context, dict):
|
|
4721
|
+
failures.append(
|
|
4722
|
+
f"final-report data.json: approval clarification `{row_id}` requires "
|
|
4723
|
+
"approvalContext under activity contract v1."
|
|
4724
|
+
)
|
|
4725
|
+
continue
|
|
4726
|
+
linked_ids = [
|
|
4727
|
+
value
|
|
4728
|
+
for value in context.get("planItemIds") or []
|
|
4729
|
+
if isinstance(value, str)
|
|
4730
|
+
]
|
|
4731
|
+
unknown_ids = set(linked_ids) - set(plan_items_by_id)
|
|
4732
|
+
if unknown_ids and not _terminal_unknown_plan_items_are_historical(
|
|
4733
|
+
row,
|
|
4734
|
+
context,
|
|
4735
|
+
unknown_ids,
|
|
4736
|
+
historical_item_ids,
|
|
4737
|
+
historical_coverage_ids,
|
|
4738
|
+
activities,
|
|
4739
|
+
):
|
|
4740
|
+
failures.append(
|
|
4741
|
+
f"final-report data.json: approval clarification `{row_id}` planItemIds "
|
|
4742
|
+
f"reference unknown plan items {sorted(unknown_ids)}."
|
|
4743
|
+
)
|
|
4744
|
+
linked_items = [
|
|
4745
|
+
plan_items_by_id[item_id]
|
|
4746
|
+
for item_id in linked_ids
|
|
4747
|
+
if item_id in plan_items_by_id
|
|
4748
|
+
]
|
|
4749
|
+
current_expected = _expected_approval_classification(
|
|
4750
|
+
row, plan_items_by_id, coverage_ids
|
|
4751
|
+
)
|
|
4752
|
+
historical_expected = _historical_approval_classification(
|
|
4753
|
+
row_id, linked_ids, historical_classes, historical_coverage_ids
|
|
4754
|
+
)
|
|
4755
|
+
expected = (
|
|
4756
|
+
historical_expected
|
|
4757
|
+
if row.get("status") in {"resolved", "obsolete"} and historical_expected
|
|
4758
|
+
else current_expected
|
|
4759
|
+
)
|
|
4760
|
+
if context.get("classification") != expected:
|
|
4761
|
+
failures.append(
|
|
4762
|
+
f"final-report data.json: approval clarification `{row_id}` classification "
|
|
4763
|
+
f"is `{context.get('classification')}` but plan evidence requires `{expected}`."
|
|
4764
|
+
)
|
|
4765
|
+
obsolete_has_active_cause = current_expected != "user-decision" or any(
|
|
4766
|
+
item_id in plan_items_by_id for item_id in linked_ids
|
|
4767
|
+
)
|
|
4768
|
+
if row.get("status") == "obsolete" and obsolete_has_active_cause:
|
|
4769
|
+
failures.append(
|
|
4770
|
+
f"final-report data.json: obsolete approval clarification `{row_id}` "
|
|
4771
|
+
f"still has an active `{current_expected}` cause in the current plan."
|
|
4772
|
+
)
|
|
4773
|
+
_validate_approval_activity_refs(row_id, context, activities, failures)
|
|
4774
|
+
_validate_approval_dispositions(row, context, failures)
|
|
4775
|
+
_validate_resolved_approval(row, context, failures)
|
|
4776
|
+
_validate_correctness_resolution(
|
|
4777
|
+
row,
|
|
4778
|
+
linked_ids,
|
|
4779
|
+
linked_items,
|
|
4780
|
+
coverage_ids,
|
|
4781
|
+
activities,
|
|
4782
|
+
approval_state,
|
|
4783
|
+
approval_state_path,
|
|
4784
|
+
activity_timestamps,
|
|
4785
|
+
failures,
|
|
4786
|
+
)
|
|
4787
|
+
if report_approved and row.get("status") in {"open", "answered"}:
|
|
4788
|
+
failures.append(
|
|
4789
|
+
f"final-report data.json: approval is true while clarification `{row_id}` "
|
|
4790
|
+
f"has status `{row.get('status')}`; open and answered approval "
|
|
4791
|
+
"rows remain blocking."
|
|
4792
|
+
)
|
|
4793
|
+
|
|
4794
|
+
|
|
4795
|
+
def _validate_activity_contract_plan_limits(
|
|
4796
|
+
data: dict,
|
|
4797
|
+
run_manifest: dict,
|
|
4798
|
+
failures: list[str],
|
|
4799
|
+
) -> None:
|
|
4800
|
+
if not _is_activity_contract_v1_planning(run_manifest):
|
|
4801
|
+
return
|
|
4802
|
+
pbv = (data.get("implementationPlanning") or {}).get("planBodyVerification") or {}
|
|
4803
|
+
rounds_applied = pbv.get("selfFixRoundsApplied", 0)
|
|
4804
|
+
if isinstance(rounds_applied, int) and rounds_applied > 1:
|
|
4805
|
+
failures.append(
|
|
4806
|
+
"final-report data.json: activity contract v1 selfFixRoundsApplied "
|
|
4807
|
+
"must be at most one automatic self-fix round"
|
|
4808
|
+
)
|
|
4809
|
+
if pbv.get("selfFixStopReason") == "cause-group-recurrence":
|
|
4810
|
+
failures.append(
|
|
4811
|
+
"final-report data.json: activity contract v1 cannot newly emit "
|
|
4812
|
+
"cause-group-recurrence"
|
|
4813
|
+
)
|
|
4814
|
+
|
|
4815
|
+
|
|
3885
4816
|
_CHECKLIST_REF_RE = re.compile(r"VC-\d+")
|
|
3886
4817
|
|
|
3887
4818
|
|
|
@@ -4172,7 +5103,11 @@ def _validate_participating_analysers(data: dict, failures: list[str]) -> None:
|
|
|
4172
5103
|
)
|
|
4173
5104
|
|
|
4174
5105
|
|
|
4175
|
-
def _validate_gate_blocked_by(
|
|
5106
|
+
def _validate_gate_blocked_by(
|
|
5107
|
+
data: dict,
|
|
5108
|
+
failures: list[str],
|
|
5109
|
+
accepted_item_ids: set[str] | None = None,
|
|
5110
|
+
) -> None:
|
|
4176
5111
|
"""The gate value names an *outcome*; `gateBlockedBy` names the *cause*.
|
|
4177
5112
|
|
|
4178
5113
|
Two independent inputs can block approval — a `majority-disagree` plan item
|
|
@@ -4200,7 +5135,12 @@ def _validate_gate_blocked_by(data: dict, failures: list[str]) -> None:
|
|
|
4200
5135
|
if isinstance(c, str) and str(c).strip()
|
|
4201
5136
|
}
|
|
4202
5137
|
coverage_blockers = _independent_coverage_blockers(ip, pbv)
|
|
4203
|
-
|
|
5138
|
+
accepted = (
|
|
5139
|
+
_resolved_noncritical_dissent_ids(data)
|
|
5140
|
+
if accepted_item_ids is None
|
|
5141
|
+
else accepted_item_ids
|
|
5142
|
+
)
|
|
5143
|
+
actual_causes = _gate_blocking_causes(pbv, coverage_blockers, accepted)
|
|
4204
5144
|
|
|
4205
5145
|
if actual_causes and declared_gate in ("passed", "passed-with-dissent"):
|
|
4206
5146
|
failures.append(
|
|
@@ -6227,7 +7167,11 @@ def _validate_plan_item_subject_substance(data: dict, failures: list[str]) -> No
|
|
|
6227
7167
|
)
|
|
6228
7168
|
|
|
6229
7169
|
|
|
6230
|
-
def _validate_plan_body_clarification_matching(
|
|
7170
|
+
def _validate_plan_body_clarification_matching(
|
|
7171
|
+
data: dict,
|
|
7172
|
+
failures: list[str],
|
|
7173
|
+
accepted_item_ids: set[str] | None = None,
|
|
7174
|
+
) -> None:
|
|
6231
7175
|
"""H5 — every plan item that the recorded verdicts make `majority-disagree`
|
|
6232
7176
|
must point (via `clarificationId`) at an existing `blocks: approval`
|
|
6233
7177
|
clarification row. Closes the hole where a majority-disagree item blocks the
|
|
@@ -6243,6 +7187,11 @@ def _validate_plan_body_clarification_matching(data: dict, failures: list[str])
|
|
|
6243
7187
|
round_count = pbv.get("roundCount")
|
|
6244
7188
|
if not isinstance(round_count, int) or round_count < 1:
|
|
6245
7189
|
return
|
|
7190
|
+
accepted = (
|
|
7191
|
+
_resolved_noncritical_dissent_ids(data)
|
|
7192
|
+
if accepted_item_ids is None
|
|
7193
|
+
else accepted_item_ids
|
|
7194
|
+
)
|
|
6246
7195
|
clar_rows = [r for r in (data.get("clarificationItems") or []) if isinstance(r, dict)]
|
|
6247
7196
|
all_ids = {r.get("id") for r in clar_rows if r.get("id")}
|
|
6248
7197
|
approval_ids = {r.get("id") for r in clar_rows if r.get("blocks") == "approval" and r.get("id")}
|
|
@@ -6251,7 +7200,7 @@ def _validate_plan_body_clarification_matching(data: dict, failures: list[str])
|
|
|
6251
7200
|
continue
|
|
6252
7201
|
if _classify_plan_item_gate(item) != "majority-disagree":
|
|
6253
7202
|
continue
|
|
6254
|
-
if _is_dissent_downgraded(item, pbv):
|
|
7203
|
+
if _is_dissent_downgraded(item, pbv, accepted):
|
|
6255
7204
|
continue
|
|
6256
7205
|
item_id = item.get("id") or "<unknown>"
|
|
6257
7206
|
cid = item.get("clarificationId")
|
|
@@ -6427,8 +7376,9 @@ def validate_plan_body_section(
|
|
|
6427
7376
|
spent against a mis-scored gate.
|
|
6428
7377
|
"""
|
|
6429
7378
|
pbv = (data.get("implementationPlanning") or {}).get("planBodyVerification") or {}
|
|
6430
|
-
|
|
6431
|
-
|
|
7379
|
+
accepted_item_ids = _resolved_noncritical_dissent_ids(data)
|
|
7380
|
+
_validate_plan_body_gate_recompute(data, failures, accepted_item_ids)
|
|
7381
|
+
_validate_gate_blocked_by(data, failures, accepted_item_ids)
|
|
6432
7382
|
_validate_participating_analysers(data, failures)
|
|
6433
7383
|
_validate_self_fix_rewrite_scope(data, failures)
|
|
6434
7384
|
_validate_self_fix_grouping(data, failures)
|
|
@@ -6440,17 +7390,21 @@ def validate_plan_body_section(
|
|
|
6440
7390
|
_validate_verdict_rounds_outlive_self_fix(data, failures)
|
|
6441
7391
|
_validate_plan_item_extraction_completeness(data, failures)
|
|
6442
7392
|
_validate_plan_item_subject_substance(data, failures)
|
|
6443
|
-
_validate_plan_body_clarification_matching(data, failures)
|
|
7393
|
+
_validate_plan_body_clarification_matching(data, failures, accepted_item_ids)
|
|
6444
7394
|
_validate_disagree_has_fixability(data, failures)
|
|
6445
7395
|
_validate_self_fix_before_clarification(data, failures)
|
|
6446
7396
|
return [*_detect_self_fix_recurrence(pbv), *_detect_uniform_verifier(pbv)]
|
|
6447
7397
|
|
|
6448
7398
|
|
|
6449
|
-
def _gate_summary_item(
|
|
7399
|
+
def _gate_summary_item(
|
|
7400
|
+
item: dict,
|
|
7401
|
+
pbv: dict,
|
|
7402
|
+
accepted_item_ids: set[str],
|
|
7403
|
+
) -> dict:
|
|
6450
7404
|
"""One `gate.items[]` row: the gate class plus its state-file counterpart."""
|
|
6451
7405
|
classification = (
|
|
6452
7406
|
"has-dissent"
|
|
6453
|
-
if _is_dissent_downgraded(item, pbv)
|
|
7407
|
+
if _is_dissent_downgraded(item, pbv, accepted_item_ids)
|
|
6454
7408
|
else _classify_plan_item_gate(item)
|
|
6455
7409
|
)
|
|
6456
7410
|
return {
|
|
@@ -6478,11 +7432,12 @@ def plan_body_gate_summary(data: dict) -> dict | None:
|
|
|
6478
7432
|
pbv = ip.get("planBodyVerification")
|
|
6479
7433
|
if not isinstance(pbv, dict):
|
|
6480
7434
|
return None
|
|
6481
|
-
|
|
7435
|
+
accepted_item_ids = _resolved_noncritical_dissent_ids(data)
|
|
7436
|
+
recomputed = _recompute_plan_body_gate(pbv, accepted_item_ids)
|
|
6482
7437
|
if recomputed is None:
|
|
6483
7438
|
return None
|
|
6484
7439
|
items = [
|
|
6485
|
-
_gate_summary_item(item, pbv)
|
|
7440
|
+
_gate_summary_item(item, pbv, accepted_item_ids)
|
|
6486
7441
|
for item in (pbv.get("planItems") or [])
|
|
6487
7442
|
if isinstance(item, dict)
|
|
6488
7443
|
]
|
|
@@ -6493,7 +7448,9 @@ def plan_body_gate_summary(data: dict) -> dict | None:
|
|
|
6493
7448
|
"declaredBlockedBy": sorted(
|
|
6494
7449
|
str(c) for c in (pbv.get("gateBlockedBy") or []) if isinstance(c, str)
|
|
6495
7450
|
),
|
|
6496
|
-
"blockedBy": sorted(
|
|
7451
|
+
"blockedBy": sorted(
|
|
7452
|
+
_gate_blocking_causes(pbv, coverage_blockers, accepted_item_ids)
|
|
7453
|
+
),
|
|
6497
7454
|
"coverageBlockers": coverage_blockers,
|
|
6498
7455
|
"blockingItems": [
|
|
6499
7456
|
item["id"] for item in items if item["classification"] == "majority-disagree"
|
|
@@ -7290,14 +8247,15 @@ def _validate_fix_cycle(run_manifest: dict, data: dict, failures: list[str]) ->
|
|
|
7290
8247
|
def _validate_session_conformance(
|
|
7291
8248
|
team_state: dict,
|
|
7292
8249
|
team_state_path: Path,
|
|
8250
|
+
run_manifest: Mapping[str, Any],
|
|
7293
8251
|
project_root: Path,
|
|
7294
8252
|
report_path: Path,
|
|
7295
8253
|
task_type: str,
|
|
7296
8254
|
claude_projects_dir: str | None,
|
|
7297
8255
|
failures: list[str],
|
|
7298
8256
|
) -> None:
|
|
7299
|
-
"""prompts/lead/okstra-lead-contract.md
|
|
7300
|
-
|
|
8257
|
+
"""prompts/lead/okstra-lead-contract.md의 PROGRESS / activity / heartbeat /
|
|
8258
|
+
implementation entry guard 사후 검사를 위임하고 실패를
|
|
7301
8259
|
``session-conformance: `` 접두로 folding 한다. 설계:
|
|
7302
8260
|
docs/superpowers/specs/2026-06-10-blocking-contract-posthoc-conformance-design.md
|
|
7303
8261
|
"""
|
|
@@ -7314,6 +8272,7 @@ def _validate_session_conformance(
|
|
|
7314
8272
|
result = validate_session_conformance(
|
|
7315
8273
|
team_state=team_state,
|
|
7316
8274
|
team_state_path=team_state_path,
|
|
8275
|
+
run_manifest=run_manifest,
|
|
7317
8276
|
project_root=project_root,
|
|
7318
8277
|
report_path=report_path,
|
|
7319
8278
|
task_type=task_type,
|
|
@@ -8149,6 +9108,7 @@ def main() -> int:
|
|
|
8149
9108
|
_validate_session_conformance(
|
|
8150
9109
|
team_state,
|
|
8151
9110
|
team_state_path,
|
|
9111
|
+
run_manifest,
|
|
8152
9112
|
project_root,
|
|
8153
9113
|
report_path,
|
|
8154
9114
|
task_type,
|