@ccoalm/ccl-skills 0.15.1 → 0.15.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +230 -16
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +2 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +12 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
- package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +6 -6
- package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/writing-judgments.md +63 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
- package/dist/assets/release.json +60 -50
- package/dist/codex-host.d.ts +1 -1
- package/dist/codex-host.js +39 -15
- package/dist/host-probe.d.ts +16 -0
- package/dist/host-probe.js +29 -5
- package/dist/operations.js +34 -8
- package/dist/unified.d.ts +1 -1
- package/dist/unified.js +11 -4
- package/package.json +1 -1
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py
CHANGED
|
@@ -419,10 +419,12 @@ CONTROLLER_OWNED_FIELDS = {
|
|
|
419
419
|
"challenge_index",
|
|
420
420
|
"challenge_rounds_remaining",
|
|
421
421
|
"completion_gated",
|
|
422
|
+
"completion_basis",
|
|
422
423
|
"completion_review_result_sha256",
|
|
423
424
|
"decision",
|
|
424
425
|
"delivery",
|
|
425
426
|
"findings_require_implementer_self_review",
|
|
427
|
+
"finding_dispositions_sha256",
|
|
426
428
|
"human_decision_required",
|
|
427
429
|
"native_skill_binding",
|
|
428
430
|
"owner_selection_evidence",
|
|
@@ -436,6 +438,7 @@ CONTROLLER_OWNED_FIELDS = {
|
|
|
436
438
|
"prior_challenge_focuses",
|
|
437
439
|
"prior_review_result_sha256",
|
|
438
440
|
"residual_risks",
|
|
441
|
+
"resolved_finding_occurrences",
|
|
439
442
|
"review_chain_id",
|
|
440
443
|
"review_chain_tracked",
|
|
441
444
|
"review_depth",
|
|
@@ -524,6 +527,11 @@ class GateError(RuntimeError):
|
|
|
524
527
|
self.reason_code = reason_code
|
|
525
528
|
|
|
526
529
|
|
|
530
|
+
class CompletionFindingsError(GateError):
|
|
531
|
+
def __init__(self, reason: str) -> None:
|
|
532
|
+
super().__init__(reason, "completion_checkpoint_invalid")
|
|
533
|
+
|
|
534
|
+
|
|
527
535
|
class GateArgumentParser(argparse.ArgumentParser):
|
|
528
536
|
def error(self, message: str) -> None:
|
|
529
537
|
raise GateError(message)
|
|
@@ -1550,8 +1558,17 @@ def _canonical_review_scope(profile: dict[str, Any]) -> dict[str, Any]:
|
|
|
1550
1558
|
return scope
|
|
1551
1559
|
|
|
1552
1560
|
|
|
1561
|
+
def _unique_json_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
1562
|
+
result: dict[str, Any] = {}
|
|
1563
|
+
for key, value in pairs:
|
|
1564
|
+
if key in result:
|
|
1565
|
+
raise ValueError(f"duplicate JSON key: {key}")
|
|
1566
|
+
result[key] = value
|
|
1567
|
+
return result
|
|
1568
|
+
|
|
1569
|
+
|
|
1553
1570
|
def _load_prior_review_result(
|
|
1554
|
-
path_value: str, expected_index: int
|
|
1571
|
+
path_value: str, expected_index: int, *, strict_json: bool = False
|
|
1555
1572
|
) -> tuple[dict[str, Any], str]:
|
|
1556
1573
|
source = Path(path_value)
|
|
1557
1574
|
if not source.is_absolute():
|
|
@@ -1571,10 +1588,13 @@ def _load_prior_review_result(
|
|
|
1571
1588
|
),
|
|
1572
1589
|
reason_code="review_chain_invalid",
|
|
1573
1590
|
)
|
|
1574
|
-
payload = json.loads(
|
|
1591
|
+
payload = json.loads(
|
|
1592
|
+
encoded.decode("utf-8"),
|
|
1593
|
+
object_pairs_hook=_unique_json_object if strict_json else None,
|
|
1594
|
+
)
|
|
1575
1595
|
except GateError:
|
|
1576
1596
|
raise
|
|
1577
|
-
except (UnicodeError,
|
|
1597
|
+
except (UnicodeError, ValueError) as exc:
|
|
1578
1598
|
raise GateError(
|
|
1579
1599
|
f"cannot read prior review result {expected_index}: {exc}",
|
|
1580
1600
|
"review_chain_invalid",
|
|
@@ -2648,7 +2668,7 @@ def freeze_review_profile(
|
|
|
2648
2668
|
or args.challenge_index
|
|
2649
2669
|
or args.review_chain_id
|
|
2650
2670
|
or args.autonomous_review_index is not None
|
|
2651
|
-
or args.prior_review_result_file
|
|
2671
|
+
or (args.prior_review_result_file and not args.finding_dispositions_file)
|
|
2652
2672
|
or args.predecessor_chain_result_file
|
|
2653
2673
|
):
|
|
2654
2674
|
raise GateError(
|
|
@@ -2660,6 +2680,11 @@ def freeze_review_profile(
|
|
|
2660
2680
|
"--completion-review-result-file is only valid in complete mode",
|
|
2661
2681
|
"completion_checkpoint_invalid",
|
|
2662
2682
|
)
|
|
2683
|
+
if args.finding_dispositions_file and args.mode != "complete":
|
|
2684
|
+
raise GateError(
|
|
2685
|
+
"--finding-dispositions-file is only valid in complete mode",
|
|
2686
|
+
"completion_checkpoint_invalid",
|
|
2687
|
+
)
|
|
2663
2688
|
risk_tags = sorted(set(args.risk_tag))
|
|
2664
2689
|
if len(risk_tags) > 20:
|
|
2665
2690
|
raise GateError("--risk-tag may be supplied at most 20 unique times")
|
|
@@ -2982,7 +3007,9 @@ def freeze_review_profile(
|
|
|
2982
3007
|
else:
|
|
2983
3008
|
if (
|
|
2984
3009
|
args.autonomous_review_index is not None
|
|
2985
|
-
or args.prior_review_result_file
|
|
3010
|
+
or (args.prior_review_result_file and not (
|
|
3011
|
+
args.mode == "complete" and args.finding_dispositions_file
|
|
3012
|
+
))
|
|
2986
3013
|
or args.predecessor_chain_result_file
|
|
2987
3014
|
):
|
|
2988
3015
|
raise GateError(
|
|
@@ -3622,10 +3649,12 @@ def validate_completion_checkpoint(
|
|
|
3622
3649
|
args: argparse.Namespace,
|
|
3623
3650
|
packet_hash: str,
|
|
3624
3651
|
profile: dict[str, Any],
|
|
3652
|
+
*,
|
|
3653
|
+
original_round: bool = False,
|
|
3625
3654
|
) -> tuple[str, dict[str, Any]]:
|
|
3626
3655
|
try:
|
|
3627
3656
|
prior, result_hash = _load_prior_review_result(
|
|
3628
|
-
args.completion_review_result_file, 1
|
|
3657
|
+
args.completion_review_result_file, 1, strict_json=original_round
|
|
3629
3658
|
)
|
|
3630
3659
|
except GateError as exc:
|
|
3631
3660
|
raise GateError(exc.reason, "completion_checkpoint_invalid") from exc
|
|
@@ -3666,6 +3695,16 @@ def validate_completion_checkpoint(
|
|
|
3666
3695
|
and isinstance(prior_gate, dict)
|
|
3667
3696
|
and prior_gate.get("required") is False
|
|
3668
3697
|
)
|
|
3698
|
+
findings_checkpoint = (
|
|
3699
|
+
prior.get("status") == "findings"
|
|
3700
|
+
and prior.get("next_action") in (
|
|
3701
|
+
"implementer_self_review", "triage_findings_and_continue_independent_work"
|
|
3702
|
+
)
|
|
3703
|
+
and isinstance(prior_gate, dict)
|
|
3704
|
+
and prior_gate.get("required") is True
|
|
3705
|
+
and isinstance(prior_gate.get("required_triggers"), list)
|
|
3706
|
+
and "findings_returned" in prior_gate["required_triggers"]
|
|
3707
|
+
)
|
|
3669
3708
|
if (
|
|
3670
3709
|
prior.get("schema_version") != 3
|
|
3671
3710
|
or prior.get("mode") not in ("review", "challenge")
|
|
@@ -3673,8 +3712,11 @@ def validate_completion_checkpoint(
|
|
|
3673
3712
|
prior.get("mode") == "challenge"
|
|
3674
3713
|
and prior.get("review_chain_tracked") is not True
|
|
3675
3714
|
)
|
|
3676
|
-
or
|
|
3677
|
-
|
|
3715
|
+
or not (
|
|
3716
|
+
(prior.get("status") == "passed" and prior.get("findings") == [])
|
|
3717
|
+
or (prior.get("status") == "findings"
|
|
3718
|
+
and isinstance(prior.get("findings"), list) and prior["findings"])
|
|
3719
|
+
)
|
|
3678
3720
|
or prior.get("candidate_sha256") != packet_hash
|
|
3679
3721
|
or prior.get("packet_sha256") != packet_hash
|
|
3680
3722
|
or prior.get("stage") != profile["stage"]
|
|
@@ -3728,15 +3770,178 @@ def validate_completion_checkpoint(
|
|
|
3728
3770
|
or prior["wording_only_proof_sha256"] is not None
|
|
3729
3771
|
or "wording_only_scope" not in prior
|
|
3730
3772
|
or prior["wording_only_scope"] is not None
|
|
3731
|
-
or not (final_round_checkpoint or early_challenge_checkpoint)
|
|
3773
|
+
or not (original_round or final_round_checkpoint or early_challenge_checkpoint or findings_checkpoint)
|
|
3732
3774
|
):
|
|
3733
3775
|
raise GateError(
|
|
3734
3776
|
"completion review result does not bind a passed exact candidate awaiting deep self-review",
|
|
3735
3777
|
"completion_checkpoint_invalid",
|
|
3736
3778
|
)
|
|
3779
|
+
if prior["status"] == "findings" and not original_round:
|
|
3780
|
+
raise CompletionFindingsError(
|
|
3781
|
+
"exact-candidate findings require complete source-refutation dispositions"
|
|
3782
|
+
)
|
|
3737
3783
|
return result_hash, prior
|
|
3738
3784
|
|
|
3739
3785
|
|
|
3786
|
+
def validate_finding_dispositions(
|
|
3787
|
+
args: argparse.Namespace,
|
|
3788
|
+
packet_hash: str,
|
|
3789
|
+
profile: dict[str, Any],
|
|
3790
|
+
) -> tuple[str, dict[str, Any], dict[str, Any]]:
|
|
3791
|
+
# These checks bind accountable local source reasoning to immutable receipt
|
|
3792
|
+
# occurrences. They validate structure and provenance, not that reasoning's truth.
|
|
3793
|
+
paths = [*args.prior_review_result_file, args.completion_review_result_file]
|
|
3794
|
+
if not 2 <= len(paths) <= profile["challenge_budget"] + 1:
|
|
3795
|
+
raise CompletionFindingsError("source refutation requires the full review-then-challenge chain")
|
|
3796
|
+
receipt_hashes: list[str] = []
|
|
3797
|
+
focuses: list[str] = []
|
|
3798
|
+
occurrences: list[dict[str, str]] = []
|
|
3799
|
+
seen_occurrences: set[tuple[str, str]] = set()
|
|
3800
|
+
chain_id = None
|
|
3801
|
+
for index, path in enumerate(paths, 1):
|
|
3802
|
+
round_args = argparse.Namespace(**vars(args))
|
|
3803
|
+
round_args.completion_review_result_file = path
|
|
3804
|
+
receipt_hash, receipt = validate_completion_checkpoint(
|
|
3805
|
+
round_args, packet_hash, profile, original_round=True
|
|
3806
|
+
)
|
|
3807
|
+
if index == 1:
|
|
3808
|
+
chain_id = receipt.get("review_chain_id")
|
|
3809
|
+
challenge_index = receipt.get("challenge_index")
|
|
3810
|
+
expected_review_state = (
|
|
3811
|
+
"post_review_budget"
|
|
3812
|
+
if receipt["status"] == "findings" and receipt["autonomous_reviews_remaining"] == 0
|
|
3813
|
+
else "findings_pending"
|
|
3814
|
+
if receipt["status"] == "findings"
|
|
3815
|
+
else "reviewed"
|
|
3816
|
+
)
|
|
3817
|
+
receipt_gate = receipt.get("self_review_gate")
|
|
3818
|
+
if (
|
|
3819
|
+
receipt.get("mode") != ("review" if index == 1 else "challenge")
|
|
3820
|
+
or receipt.get("review_chain_tracked") is not True
|
|
3821
|
+
or receipt.get("review_chain_id") != chain_id
|
|
3822
|
+
or receipt.get("autonomous_review_index") != index
|
|
3823
|
+
or type(challenge_index) is not int
|
|
3824
|
+
or challenge_index != index - 1
|
|
3825
|
+
or receipt.get("challenge_rounds_remaining") != profile["challenge_budget"] - challenge_index
|
|
3826
|
+
or receipt.get("prior_review_result_sha256") != receipt_hashes
|
|
3827
|
+
or receipt.get("prior_challenge_focuses") != focuses
|
|
3828
|
+
or receipt.get("review_state") != expected_review_state
|
|
3829
|
+
or receipt.get("human_decision_required") is not (expected_review_state == "post_review_budget")
|
|
3830
|
+
or (
|
|
3831
|
+
receipt["status"] == "findings"
|
|
3832
|
+
and (
|
|
3833
|
+
not isinstance(receipt_gate, dict)
|
|
3834
|
+
or receipt_gate.get("required") is not True
|
|
3835
|
+
or not isinstance(receipt_gate.get("required_triggers"), list)
|
|
3836
|
+
or "findings_returned" not in receipt_gate["required_triggers"]
|
|
3837
|
+
)
|
|
3838
|
+
)
|
|
3839
|
+
or any(receipt.get(key) is not None for key in (
|
|
3840
|
+
"predecessor_chain_id", "predecessor_result_sha256", "predecessor_candidate_sha256"
|
|
3841
|
+
))
|
|
3842
|
+
or receipt.get("predecessor_challenge_focuses") not in (None, [])
|
|
3843
|
+
):
|
|
3844
|
+
raise CompletionFindingsError("source refutation requires original contiguous normal-chain receipts")
|
|
3845
|
+
focus = receipt.get("challenge_focus")
|
|
3846
|
+
if index > 1:
|
|
3847
|
+
if not isinstance(focus, str) or not focus.strip() or focus in focuses:
|
|
3848
|
+
raise CompletionFindingsError("source refutation requires distinct recorded challenge focuses")
|
|
3849
|
+
focuses.append(focus)
|
|
3850
|
+
elif focus is not None:
|
|
3851
|
+
raise CompletionFindingsError("initial review cannot carry a challenge focus")
|
|
3852
|
+
for finding in receipt["findings"]:
|
|
3853
|
+
if (
|
|
3854
|
+
not isinstance(finding, dict)
|
|
3855
|
+
or finding.get("severity") not in ("P0", "P1", "P2")
|
|
3856
|
+
or type(finding.get("line")) is not int
|
|
3857
|
+
or finding["line"] < 1
|
|
3858
|
+
or any(not isinstance(finding.get(key), str) or not finding[key].strip()
|
|
3859
|
+
for key in ("file", "failure_path", "smallest_fix"))
|
|
3860
|
+
):
|
|
3861
|
+
raise CompletionFindingsError("original receipt contains an invalid finding")
|
|
3862
|
+
try:
|
|
3863
|
+
finding_hash = _canonical_digest(finding)
|
|
3864
|
+
except (UnicodeError, ValueError) as exc:
|
|
3865
|
+
raise CompletionFindingsError("original finding must be canonical UTF-8 JSON") from exc
|
|
3866
|
+
pair = (receipt_hash, finding_hash)
|
|
3867
|
+
if pair in seen_occurrences:
|
|
3868
|
+
# Identical canonical content shares one occurrence identity;
|
|
3869
|
+
# preserve every entry in the original receipt unchanged.
|
|
3870
|
+
continue
|
|
3871
|
+
seen_occurrences.add(pair)
|
|
3872
|
+
occurrences.append({"receipt_sha256": receipt_hash, "finding_sha256": finding_hash})
|
|
3873
|
+
receipt_hashes.append(receipt_hash)
|
|
3874
|
+
if not occurrences:
|
|
3875
|
+
raise CompletionFindingsError("source refutation requires at least one original finding")
|
|
3876
|
+
source = Path(args.finding_dispositions_file)
|
|
3877
|
+
if not source.is_absolute():
|
|
3878
|
+
raise CompletionFindingsError("finding dispositions path must be absolute")
|
|
3879
|
+
try:
|
|
3880
|
+
encoded = read_bounded_regular_file(
|
|
3881
|
+
source,
|
|
3882
|
+
label="finding dispositions",
|
|
3883
|
+
maximum=MAX_RESULT_BYTES,
|
|
3884
|
+
regular_error="finding dispositions must be a bounded regular JSON file",
|
|
3885
|
+
oversized_error="finding dispositions exceed the size limit",
|
|
3886
|
+
reason_code="completion_checkpoint_invalid",
|
|
3887
|
+
)
|
|
3888
|
+
manifest = json.loads(encoded.decode("utf-8"), object_pairs_hook=_unique_json_object)
|
|
3889
|
+
except GateError:
|
|
3890
|
+
raise
|
|
3891
|
+
except (UnicodeError, ValueError) as exc:
|
|
3892
|
+
raise CompletionFindingsError(f"cannot read finding dispositions: {exc}") from exc
|
|
3893
|
+
if (
|
|
3894
|
+
not isinstance(manifest, dict)
|
|
3895
|
+
or set(manifest) != {"schema_version", "candidate_sha256", "review_result_sha256", "dispositions"}
|
|
3896
|
+
or type(manifest.get("schema_version")) is not int
|
|
3897
|
+
or manifest["schema_version"] != 1
|
|
3898
|
+
or manifest.get("candidate_sha256") != packet_hash
|
|
3899
|
+
or manifest.get("review_result_sha256") != receipt_hashes
|
|
3900
|
+
or not isinstance(manifest.get("dispositions"), list)
|
|
3901
|
+
or len(manifest["dispositions"]) != len(occurrences)
|
|
3902
|
+
):
|
|
3903
|
+
raise CompletionFindingsError("finding dispositions do not bind the exact candidate and full original chain")
|
|
3904
|
+
for disposition, occurrence in zip(manifest["dispositions"], occurrences):
|
|
3905
|
+
if (
|
|
3906
|
+
not isinstance(disposition, dict)
|
|
3907
|
+
or set(disposition) != {"receipt_sha256", "finding_sha256", "disposition", "evidence"}
|
|
3908
|
+
or disposition.get("receipt_sha256") != occurrence["receipt_sha256"]
|
|
3909
|
+
or disposition.get("finding_sha256") != occurrence["finding_sha256"]
|
|
3910
|
+
or disposition.get("disposition") != "source_refuted"
|
|
3911
|
+
):
|
|
3912
|
+
raise CompletionFindingsError("finding dispositions must cover every original occurrence once in order")
|
|
3913
|
+
evidence = disposition.get("evidence")
|
|
3914
|
+
if (
|
|
3915
|
+
not isinstance(evidence, list)
|
|
3916
|
+
or not evidence
|
|
3917
|
+
or any(
|
|
3918
|
+
not isinstance(item, str) or item != item.strip() or not 1 <= len(item) <= 1000
|
|
3919
|
+
for item in evidence
|
|
3920
|
+
)
|
|
3921
|
+
or len(set(evidence)) != len(evidence)
|
|
3922
|
+
):
|
|
3923
|
+
raise CompletionFindingsError("each source refutation requires nonempty bounded distinct evidence")
|
|
3924
|
+
if any(
|
|
3925
|
+
ord(char) < 0x20
|
|
3926
|
+
or 0x7F <= ord(char) <= 0x9F
|
|
3927
|
+
or char in "\u2028\u2029"
|
|
3928
|
+
or unicodedata.category(char) == "Cf"
|
|
3929
|
+
for item in evidence
|
|
3930
|
+
for char in item
|
|
3931
|
+
):
|
|
3932
|
+
raise CompletionFindingsError("source-refutation evidence must not contain control or format characters")
|
|
3933
|
+
try:
|
|
3934
|
+
for item in evidence:
|
|
3935
|
+
item.encode("utf-8")
|
|
3936
|
+
except UnicodeError as exc:
|
|
3937
|
+
raise CompletionFindingsError("source-refutation evidence must be valid UTF-8 text") from exc
|
|
3938
|
+
return receipt_hashes[-1], receipt, {
|
|
3939
|
+
"completion_basis": "source_refuted_findings",
|
|
3940
|
+
"finding_dispositions_sha256": hashlib.sha256(encoded).hexdigest(),
|
|
3941
|
+
"resolved_finding_occurrences": occurrences,
|
|
3942
|
+
}
|
|
3943
|
+
|
|
3944
|
+
|
|
3740
3945
|
def record_skip(
|
|
3741
3946
|
result: dict[str, Any],
|
|
3742
3947
|
client: str,
|
|
@@ -3875,6 +4080,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
3875
4080
|
parser.add_argument("--prior-review-result-file", action="append", default=[])
|
|
3876
4081
|
parser.add_argument("--predecessor-chain-result-file", default=None)
|
|
3877
4082
|
parser.add_argument("--completion-review-result-file")
|
|
4083
|
+
parser.add_argument("--finding-dispositions-file")
|
|
3878
4084
|
parser.add_argument("--allow-fallback-egress", action="store_true")
|
|
3879
4085
|
parser.add_argument("--host-remediation-attempted", action="store_true")
|
|
3880
4086
|
parser.add_argument("--review-harness", action="store_true")
|
|
@@ -3991,24 +4197,30 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
3991
4197
|
)
|
|
3992
4198
|
if args.mode == "complete":
|
|
3993
4199
|
try:
|
|
3994
|
-
|
|
3995
|
-
completion_result_hash,
|
|
3996
|
-
|
|
3997
|
-
|
|
4200
|
+
if args.finding_dispositions_file:
|
|
4201
|
+
completion_result_hash, completed_review, completion_metadata = validate_finding_dispositions(
|
|
4202
|
+
args, packet_hash, profile
|
|
4203
|
+
)
|
|
4204
|
+
else:
|
|
4205
|
+
completion_result_hash, completed_review = validate_completion_checkpoint(
|
|
4206
|
+
args, packet_hash, profile
|
|
4207
|
+
)
|
|
4208
|
+
completion_metadata = {"completion_basis": "external_pass"}
|
|
3998
4209
|
except GateError as exc:
|
|
4210
|
+
resolve_findings = isinstance(exc, CompletionFindingsError) or bool(args.finding_dispositions_file)
|
|
3999
4211
|
result.update(
|
|
4000
4212
|
reason=exc.reason,
|
|
4001
4213
|
reason_code=exc.reason_code,
|
|
4002
|
-
next_action="run_external_review_for_current_candidate",
|
|
4214
|
+
next_action=("resolve_review_findings" if resolve_findings else "run_external_review_for_current_candidate"),
|
|
4003
4215
|
review_state="self_reviewing",
|
|
4004
4216
|
self_review_gate=self_review_gate(
|
|
4005
|
-
required_triggers=["material_candidate_change"],
|
|
4217
|
+
required_triggers=(["findings_returned", "before_completion_claim"] if resolve_findings else ["material_candidate_change"]),
|
|
4006
4218
|
satisfied_triggers=profile["self_review_satisfied_triggers"],
|
|
4007
4219
|
blocks=["completion_claim"],
|
|
4008
4220
|
allowed_next_actions=[
|
|
4009
4221
|
"deep_self_review",
|
|
4010
4222
|
"continue_implementation",
|
|
4011
|
-
"run_external_review_after_self_review",
|
|
4223
|
+
"resolve_review_findings" if resolve_findings else "run_external_review_after_self_review",
|
|
4012
4224
|
],
|
|
4013
4225
|
),
|
|
4014
4226
|
)
|
|
@@ -4036,6 +4248,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
4036
4248
|
satisfied_triggers=["before_completion_claim"]
|
|
4037
4249
|
),
|
|
4038
4250
|
)
|
|
4251
|
+
result.update(completion_metadata)
|
|
4039
4252
|
return emit(result, 0)
|
|
4040
4253
|
last_reason_code = "no_independent_reviewer_available"
|
|
4041
4254
|
for client in order:
|
|
@@ -4271,6 +4484,7 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
4271
4484
|
"post_review_budget_checkpoint"
|
|
4272
4485
|
)
|
|
4273
4486
|
allowed_self_review_actions.append("continue_independent_work")
|
|
4487
|
+
allowed_self_review_actions.append("resolve_review_findings")
|
|
4274
4488
|
current_self_review_gate = self_review_gate(
|
|
4275
4489
|
required_triggers=required_self_review_triggers,
|
|
4276
4490
|
satisfied_triggers=profile["self_review_satisfied_triggers"],
|
|
@@ -652,7 +652,7 @@ cat >"$WORK/bin/codex" <<'CODEX_STUB'
|
|
|
652
652
|
#!/usr/bin/env bash
|
|
653
653
|
set -u
|
|
654
654
|
state="$REVIEW_WRAPPER_TEST_STATE"
|
|
655
|
-
if [ "${1:-}" = exec ] && [ "
|
|
655
|
+
if [ "${1:-}" = exec ] && [[ " $* " = *" --help "* ]]; then
|
|
656
656
|
if [ "${STUB_BEHAVIOR:-}" = help_hang ]; then
|
|
657
657
|
trap '' TERM
|
|
658
658
|
while :; do /bin/sleep 1; done
|
|
@@ -667,16 +667,72 @@ if [ "${1:-}" = exec ] && [ "${2:-}" = --disable ] && [ "${3:-}" = hooks ] && [
|
|
|
667
667
|
fi
|
|
668
668
|
if [ "${1:-}" = features ] && [ "${2:-}" = list ]; then
|
|
669
669
|
touch "$state/codex_features_invoked"
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
670
|
+
hooks_enabled=true
|
|
671
|
+
shell_enabled=true
|
|
672
|
+
while [ "$#" -gt 0 ]; do
|
|
673
|
+
case "$1" in
|
|
674
|
+
--disable)
|
|
675
|
+
case "${2:-}" in hooks) hooks_enabled=false ;; shell_tool) shell_enabled=false ;; esac
|
|
676
|
+
shift 2 ;;
|
|
677
|
+
*) shift ;;
|
|
678
|
+
esac
|
|
679
|
+
done
|
|
680
|
+
case "${STUB_BEHAVIOR:-pass}" in
|
|
681
|
+
missing_hooks_feature) ;;
|
|
682
|
+
removed_hooks_feature) printf '%s\n' 'hooks removed false' ;;
|
|
683
|
+
*) printf 'hooks stable %s\n' "$hooks_enabled" ;;
|
|
684
|
+
esac
|
|
685
|
+
case "${STUB_BEHAVIOR:-pass}" in
|
|
686
|
+
missing_shell_feature) ;;
|
|
687
|
+
removed_shell_feature) printf '%s\n' 'shell_tool removed false' ;;
|
|
688
|
+
ignored_shell_disable) printf '%s\n' 'shell_tool stable true' ;;
|
|
689
|
+
*) printf 'shell_tool stable %s\n' "$shell_enabled" ;;
|
|
690
|
+
esac
|
|
691
|
+
exit 0
|
|
692
|
+
fi
|
|
693
|
+
if [ "${1:-}" = mcp ] && [ "${2:-}" = list ]; then
|
|
694
|
+
touch "$state/codex_mcp_probe_invoked"
|
|
695
|
+
if [ "${STUB_BEHAVIOR:-pass}" = mcp_capability_missing ]; then
|
|
696
|
+
printf '%s\n' '[]'
|
|
676
697
|
exit 0
|
|
677
698
|
fi
|
|
678
|
-
|
|
679
|
-
|
|
699
|
+
python3 - "$@" <<'PY_MCP_LIST'
|
|
700
|
+
import json, os, sys, tomllib
|
|
701
|
+
|
|
702
|
+
arguments = iter(sys.argv[1:])
|
|
703
|
+
servers = {}
|
|
704
|
+
inherited_names = {
|
|
705
|
+
"inherited_mcp": "unrelated",
|
|
706
|
+
"inherited_mcp_dot": "unrelated.name",
|
|
707
|
+
"inherited_mcp_space": "unrelated name",
|
|
708
|
+
"inherited_mcp_quote": 'unrelated"name',
|
|
709
|
+
}
|
|
710
|
+
inherited_name = inherited_names.get(os.environ.get("STUB_BEHAVIOR"))
|
|
711
|
+
if inherited_name is not None:
|
|
712
|
+
servers[inherited_name] = {"command": "/bin/false", "args": [], "enabled": True}
|
|
713
|
+
for argument in arguments:
|
|
714
|
+
if argument in {"-c", "--config"}:
|
|
715
|
+
# Codex splits the override path separately from its TOML value;
|
|
716
|
+
# quotes in a dotted left-hand key are not TOML key quoting.
|
|
717
|
+
key, value = next(arguments).split("=", 1)
|
|
718
|
+
decoded = tomllib.loads("value=" + value)["value"]
|
|
719
|
+
if key == "mcp_servers":
|
|
720
|
+
configured_servers = decoded
|
|
721
|
+
elif key.startswith("mcp_servers."):
|
|
722
|
+
_, name, field = key.split(".")
|
|
723
|
+
configured_servers = {name: {field: decoded}}
|
|
724
|
+
else:
|
|
725
|
+
continue
|
|
726
|
+
for name, settings in configured_servers.items():
|
|
727
|
+
servers.setdefault(name, {}).update(settings)
|
|
728
|
+
print(json.dumps([
|
|
729
|
+
{"name": name, "enabled": server.get("enabled", True),
|
|
730
|
+
"transport": {"type": "stdio", "command": server["command"],
|
|
731
|
+
"args": server.get("args", [])}}
|
|
732
|
+
for name, server in servers.items()
|
|
733
|
+
]))
|
|
734
|
+
PY_MCP_LIST
|
|
735
|
+
exit $?
|
|
680
736
|
fi
|
|
681
737
|
touch "$state/codex_invoked"
|
|
682
738
|
printf '%s' "$0" >"$state/codex_argv0"
|
|
@@ -687,14 +743,19 @@ has_read_only=no
|
|
|
687
743
|
has_ephemeral=no
|
|
688
744
|
has_ignore_rules=no
|
|
689
745
|
has_hooks_disabled=no
|
|
746
|
+
has_shell_disabled=no
|
|
690
747
|
workspace=""
|
|
748
|
+
: >"$state/codex_configs"
|
|
691
749
|
while [ "$#" -gt 0 ]; do
|
|
692
750
|
case "$1" in
|
|
693
751
|
--output-last-message) last_message="$2"; shift 2 ;;
|
|
694
752
|
--model|-m) has_model=yes; shift 2 ;;
|
|
695
753
|
--sandbox) [ "$2" = read-only ] && has_read_only=yes; shift 2 ;;
|
|
696
754
|
--ephemeral) has_ephemeral=yes; shift ;;
|
|
697
|
-
--disable)
|
|
755
|
+
--disable)
|
|
756
|
+
case "$2" in hooks) has_hooks_disabled=yes ;; shell_tool) has_shell_disabled=yes ;; esac
|
|
757
|
+
shift 2 ;;
|
|
758
|
+
-c|--config) printf '%s\n' "$2" >>"$state/codex_configs"; shift 2 ;;
|
|
698
759
|
--ignore-rules) has_ignore_rules=yes; shift ;;
|
|
699
760
|
-C) workspace="$2"; shift 2 ;;
|
|
700
761
|
*) shift ;;
|
|
@@ -705,6 +766,7 @@ printf '%s' "$has_read_only" >"$state/codex_read_only"
|
|
|
705
766
|
printf '%s' "$has_ephemeral" >"$state/codex_ephemeral"
|
|
706
767
|
printf '%s' "$has_ignore_rules" >"$state/codex_ignore_rules"
|
|
707
768
|
printf '%s' "$has_hooks_disabled" >"$state/codex_hooks_disabled"
|
|
769
|
+
printf '%s' "$has_shell_disabled" >"$state/codex_shell_disabled"
|
|
708
770
|
[ -z "$workspace" ] || printf '%s' "$workspace" >"$state/codex_workspace"
|
|
709
771
|
if [ -n "$workspace" ] && [ -L "$workspace/.agents/skills/testing-strategy" ]; then
|
|
710
772
|
readlink "$workspace/.agents/skills/testing-strategy" >"$state/codex_skill_link"
|
|
@@ -741,7 +803,8 @@ fi
|
|
|
741
803
|
printf '%s\n' '{"type":"thread.started","thread_id":"test-thread"}'
|
|
742
804
|
printf '%s\n' '{"type":"turn.started"}'
|
|
743
805
|
case "$behavior" in
|
|
744
|
-
|
|
806
|
+
packet_read|packet_search|packet_tampered) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
807
|
+
pass|shell_disable_required|missing_shell_feature|removed_shell_feature|ignored_shell_disable|mcp_capability_missing|inherited_mcp|inherited_mcp_dot|inherited_mcp_space|inherited_mcp_quote|skills_budget_warning|skills_budget_warning_after_concern|hook_trust_warning|hook_trust_warning_started|hook_trust_warning_repeated_after_concern|unknown_error_valid_result) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
745
808
|
legacy_pass) printf '%s\n' '{"status":"passed","findings":[]}' >"$last_message" ;;
|
|
746
809
|
stream_gap) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
747
810
|
tool) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
|
|
@@ -750,9 +813,46 @@ case "$behavior" in
|
|
|
750
813
|
invalid) printf '%s\n' 'not-json' >"$last_message" ;;
|
|
751
814
|
invalid_concern) printf '%s\n' 'P1 src/example.py:7 concern without valid JSON' >"$last_message" ;;
|
|
752
815
|
esac
|
|
816
|
+
if [[ "$behavior" = packet_* ]]; then
|
|
817
|
+
python3 - "$state" "$behavior" <<'PY_PACKET_CALL'
|
|
818
|
+
import json, subprocess, sys, tomllib
|
|
819
|
+
from pathlib import Path
|
|
820
|
+
|
|
821
|
+
state, behavior = Path(sys.argv[1]), sys.argv[2]
|
|
822
|
+
servers = {}
|
|
823
|
+
for override in (state / "codex_configs").read_text().splitlines():
|
|
824
|
+
key, value = override.split("=", 1)
|
|
825
|
+
if key == "mcp_servers":
|
|
826
|
+
servers = tomllib.loads("servers=" + value)["servers"]
|
|
827
|
+
server = servers["code_review_packet"]
|
|
828
|
+
tool = "search_packet" if behavior == "packet_search" else "read_packet"
|
|
829
|
+
arguments = ({"query": "diff --git", "byte_offset": 0, "limit": 1}
|
|
830
|
+
if tool == "search_packet" else {"byte_offset": 0, "max_bytes": 46000})
|
|
831
|
+
request = {"jsonrpc": "2.0", "id": 1, "method": "tools/call",
|
|
832
|
+
"params": {"name": tool, "arguments": arguments}}
|
|
833
|
+
response = subprocess.run([server["command"], *server["args"]],
|
|
834
|
+
input=json.dumps(request) + "\n", text=True,
|
|
835
|
+
capture_output=True, timeout=5, check=True)
|
|
836
|
+
result = json.loads(response.stdout)["result"]
|
|
837
|
+
assert not result.get("isError"), result
|
|
838
|
+
if behavior == "packet_tampered":
|
|
839
|
+
result["content"][0]["text"] += "forged packet bytes"
|
|
840
|
+
item = {"type": "mcp_tool_call", "id": "packet-call-1",
|
|
841
|
+
"server": "code_review_packet", "tool": tool, "arguments": arguments,
|
|
842
|
+
"status": "in_progress"}
|
|
843
|
+
print(json.dumps({"type": "item.started", "item": item}))
|
|
844
|
+
item.update(status="completed", result=result)
|
|
845
|
+
print(json.dumps({"type": "item.completed", "item": item}))
|
|
846
|
+
(state / "codex_packet_call").write_text(tool)
|
|
847
|
+
PY_PACKET_CALL
|
|
848
|
+
[ "$?" = 0 ] || exit 42
|
|
849
|
+
fi
|
|
753
850
|
if [ "$behavior" = stream_gap ]; then
|
|
754
851
|
printf '%s\n' '{"type":"item.completed","item":{"type":"error","message":"in-process app-server event stream lagged; dropped 2 events"}}'
|
|
755
852
|
fi
|
|
853
|
+
if [ "$behavior" = shell_disable_required ] && [ "$has_shell_disabled" != yes ]; then
|
|
854
|
+
printf '%s\n' '{"type":"item.completed","item":{"type":"command_execution","command":"pwd"}}'
|
|
855
|
+
fi
|
|
756
856
|
if [ "$behavior" = tool ]; then
|
|
757
857
|
printf '%s\n' '{"type":"item.completed","item":{"type":"command_execution","command":"pwd"}}'
|
|
758
858
|
elif [ "$behavior" = foreign_tool ]; then
|
|
@@ -1676,6 +1776,60 @@ out="$(run_codex removed_hooks_feature)"; rc=$?
|
|
|
1676
1776
|
check "Codex with a removed hooks feature key falls back before inference" \
|
|
1677
1777
|
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_hook_disable_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
|
|
1678
1778
|
|
|
1779
|
+
out="$(run_codex shell_disable_required)"; rc=$?
|
|
1780
|
+
check "Codex disables shell execution before the reviewer can request a command" \
|
|
1781
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(cat "$WORK/state/codex_shell_disabled")" = yes ]'
|
|
1782
|
+
|
|
1783
|
+
for shell_capability in missing_shell_feature removed_shell_feature ignored_shell_disable; do
|
|
1784
|
+
rm -f "$WORK/state/codex_invoked"
|
|
1785
|
+
out="$(run_codex "$shell_capability")"; rc=$?
|
|
1786
|
+
check "Codex $shell_capability falls back before inference" \
|
|
1787
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_shell_disable_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
|
|
1788
|
+
done
|
|
1789
|
+
|
|
1790
|
+
rm -f "$WORK/state/codex_invoked"
|
|
1791
|
+
out="$(run_codex mcp_capability_missing)"; rc=$?
|
|
1792
|
+
check "Codex without the bounded packet MCP server falls back before inference" \
|
|
1793
|
+
'[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_packet_tools_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
|
|
1794
|
+
|
|
1795
|
+
for packet_tool in read search; do
|
|
1796
|
+
rm -f "$WORK/state/codex_packet_call"
|
|
1797
|
+
out="$(run_codex "packet_$packet_tool")"; rc=$?
|
|
1798
|
+
check "Codex accepts a real frozen packet $packet_tool through wrapper and parser" \
|
|
1799
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(cat "$WORK/state/codex_packet_call")" = "${packet_tool}_packet" ]'
|
|
1800
|
+
done
|
|
1801
|
+
out="$(run_codex packet_tampered)"; rc=$?
|
|
1802
|
+
check "Codex rejects altered packet tool bytes through wrapper and parser" \
|
|
1803
|
+
'[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = binding_mismatch ] && [ "$(field cascade_eligible "$out")" = False ]'
|
|
1804
|
+
|
|
1805
|
+
for inherited_mcp_case in inherited_mcp inherited_mcp_dot inherited_mcp_space inherited_mcp_quote; do
|
|
1806
|
+
case "$inherited_mcp_case" in
|
|
1807
|
+
inherited_mcp) inherited_mcp_name=unrelated ;;
|
|
1808
|
+
inherited_mcp_dot) inherited_mcp_name=unrelated.name ;;
|
|
1809
|
+
inherited_mcp_space) inherited_mcp_name='unrelated name' ;;
|
|
1810
|
+
inherited_mcp_quote) inherited_mcp_name='unrelated"name' ;;
|
|
1811
|
+
esac
|
|
1812
|
+
rm -f "$WORK/state/codex_invoked" "$WORK/state/codex_configs"
|
|
1813
|
+
out="$(run_codex "$inherited_mcp_case")"; rc=$?
|
|
1814
|
+
inherited_mcp_disabled="$(python3 - "$inherited_mcp_name" "$WORK/state/codex_configs" <<'PY_MCP_DISABLED'
|
|
1815
|
+
import sys, tomllib
|
|
1816
|
+
from pathlib import Path
|
|
1817
|
+
|
|
1818
|
+
path = Path(sys.argv[2])
|
|
1819
|
+
tables = []
|
|
1820
|
+
for override in path.read_text().splitlines() if path.exists() else []:
|
|
1821
|
+
key, value = override.split("=", 1)
|
|
1822
|
+
if key.strip() == "mcp_servers":
|
|
1823
|
+
tables.append(tomllib.loads("servers=" + value)["servers"])
|
|
1824
|
+
print(len(tables) == 1
|
|
1825
|
+
and tables[0].get(sys.argv[1], {}).get("enabled") is False
|
|
1826
|
+
and "code_review_packet" in tables[0])
|
|
1827
|
+
PY_MCP_DISABLED
|
|
1828
|
+
)"
|
|
1829
|
+
check "Codex encodes and disables the inherited MCP key ($inherited_mcp_case)" \
|
|
1830
|
+
'[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ -e "$WORK/state/codex_invoked" ] && [ "$inherited_mcp_disabled" = True ]'
|
|
1831
|
+
done
|
|
1832
|
+
|
|
1679
1833
|
rm -f "$WORK/state/codex_invoked" "$WORK/state/codex_help_invoked"
|
|
1680
1834
|
probe_started=$SECONDS
|
|
1681
1835
|
out="$(run_codex help_hang)"; rc=$?
|