@ccoalm/ccl-skills 0.15.1 → 0.15.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +3 -1
  2. package/dist/assets/marketplace/plugins/ccl-skills/agent-context/session-start.md +1 -1
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/app-cross-platform-dev/SKILL.md +2 -2
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +6 -5
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/development-completion.md +26 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +37 -13
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +5 -2
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/codex_review.sh +77 -5
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/kimi_packet_mcp.py +98 -4
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_cli_review.py +48 -1
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +230 -16
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_cli_review_wrappers.sh +165 -11
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_kimi_packet_mcp.py +143 -0
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_review_client_compat.py +572 -0
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +3 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/go-microservice-dev/SKILL.md +1 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/llm-inference-integration/SKILL.md +2 -0
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/multi-agent-delegation/SKILL.md +1 -1
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/nodejs-service-dev/SKILL.md +2 -0
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +3 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +3 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-service-connectivity/SKILL.md +2 -0
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +7 -7
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/design-review-gate-mechanics.md +1 -1
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/pre-final-continuation-gate.md +20 -11
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/refactoring-discipline.md +7 -1
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/python-service-dev/SKILL.md +1 -1
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +2 -2
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +17 -17
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +4 -4
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/resume-paused-delivery.md +3 -3
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +12 -0
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_ai_coding_implementation_gates.sh +83 -48
  35. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_body_compliance_grading.sh +80 -2
  36. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_controlled_escalation_pins.sh +3 -2
  37. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_validate_extraction_review_state.sh +190 -2
  38. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/validate_extraction_review_state.py +106 -4
  39. package/dist/assets/marketplace/plugins/ccl-skills/skills/terminal-cli-dev/SKILL.md +1 -1
  40. package/dist/assets/marketplace/plugins/ccl-skills/skills/testing-strategy/SKILL.md +3 -1
  41. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/SKILL.md +6 -6
  42. package/dist/assets/marketplace/plugins/ccl-skills/skills/tighten-doc/references/writing-judgments.md +63 -0
  43. package/dist/assets/marketplace/plugins/ccl-skills/skills/web-react-dev/SKILL.md +1 -1
  44. package/dist/assets/release.json +60 -50
  45. package/dist/codex-host.d.ts +1 -1
  46. package/dist/codex-host.js +39 -15
  47. package/dist/host-probe.d.ts +16 -0
  48. package/dist/host-probe.js +29 -5
  49. package/dist/operations.js +34 -8
  50. package/dist/unified.d.ts +1 -1
  51. package/dist/unified.js +11 -4
  52. package/package.json +1 -1
@@ -419,10 +419,12 @@ CONTROLLER_OWNED_FIELDS = {
419
419
  "challenge_index",
420
420
  "challenge_rounds_remaining",
421
421
  "completion_gated",
422
+ "completion_basis",
422
423
  "completion_review_result_sha256",
423
424
  "decision",
424
425
  "delivery",
425
426
  "findings_require_implementer_self_review",
427
+ "finding_dispositions_sha256",
426
428
  "human_decision_required",
427
429
  "native_skill_binding",
428
430
  "owner_selection_evidence",
@@ -436,6 +438,7 @@ CONTROLLER_OWNED_FIELDS = {
436
438
  "prior_challenge_focuses",
437
439
  "prior_review_result_sha256",
438
440
  "residual_risks",
441
+ "resolved_finding_occurrences",
439
442
  "review_chain_id",
440
443
  "review_chain_tracked",
441
444
  "review_depth",
@@ -524,6 +527,11 @@ class GateError(RuntimeError):
524
527
  self.reason_code = reason_code
525
528
 
526
529
 
530
+ class CompletionFindingsError(GateError):
531
+ def __init__(self, reason: str) -> None:
532
+ super().__init__(reason, "completion_checkpoint_invalid")
533
+
534
+
527
535
  class GateArgumentParser(argparse.ArgumentParser):
528
536
  def error(self, message: str) -> None:
529
537
  raise GateError(message)
@@ -1550,8 +1558,17 @@ def _canonical_review_scope(profile: dict[str, Any]) -> dict[str, Any]:
1550
1558
  return scope
1551
1559
 
1552
1560
 
1561
+ def _unique_json_object(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
1562
+ result: dict[str, Any] = {}
1563
+ for key, value in pairs:
1564
+ if key in result:
1565
+ raise ValueError(f"duplicate JSON key: {key}")
1566
+ result[key] = value
1567
+ return result
1568
+
1569
+
1553
1570
  def _load_prior_review_result(
1554
- path_value: str, expected_index: int
1571
+ path_value: str, expected_index: int, *, strict_json: bool = False
1555
1572
  ) -> tuple[dict[str, Any], str]:
1556
1573
  source = Path(path_value)
1557
1574
  if not source.is_absolute():
@@ -1571,10 +1588,13 @@ def _load_prior_review_result(
1571
1588
  ),
1572
1589
  reason_code="review_chain_invalid",
1573
1590
  )
1574
- payload = json.loads(encoded.decode("utf-8"))
1591
+ payload = json.loads(
1592
+ encoded.decode("utf-8"),
1593
+ object_pairs_hook=_unique_json_object if strict_json else None,
1594
+ )
1575
1595
  except GateError:
1576
1596
  raise
1577
- except (UnicodeError, json.JSONDecodeError) as exc:
1597
+ except (UnicodeError, ValueError) as exc:
1578
1598
  raise GateError(
1579
1599
  f"cannot read prior review result {expected_index}: {exc}",
1580
1600
  "review_chain_invalid",
@@ -2648,7 +2668,7 @@ def freeze_review_profile(
2648
2668
  or args.challenge_index
2649
2669
  or args.review_chain_id
2650
2670
  or args.autonomous_review_index is not None
2651
- or args.prior_review_result_file
2671
+ or (args.prior_review_result_file and not args.finding_dispositions_file)
2652
2672
  or args.predecessor_chain_result_file
2653
2673
  ):
2654
2674
  raise GateError(
@@ -2660,6 +2680,11 @@ def freeze_review_profile(
2660
2680
  "--completion-review-result-file is only valid in complete mode",
2661
2681
  "completion_checkpoint_invalid",
2662
2682
  )
2683
+ if args.finding_dispositions_file and args.mode != "complete":
2684
+ raise GateError(
2685
+ "--finding-dispositions-file is only valid in complete mode",
2686
+ "completion_checkpoint_invalid",
2687
+ )
2663
2688
  risk_tags = sorted(set(args.risk_tag))
2664
2689
  if len(risk_tags) > 20:
2665
2690
  raise GateError("--risk-tag may be supplied at most 20 unique times")
@@ -2982,7 +3007,9 @@ def freeze_review_profile(
2982
3007
  else:
2983
3008
  if (
2984
3009
  args.autonomous_review_index is not None
2985
- or args.prior_review_result_file
3010
+ or (args.prior_review_result_file and not (
3011
+ args.mode == "complete" and args.finding_dispositions_file
3012
+ ))
2986
3013
  or args.predecessor_chain_result_file
2987
3014
  ):
2988
3015
  raise GateError(
@@ -3622,10 +3649,12 @@ def validate_completion_checkpoint(
3622
3649
  args: argparse.Namespace,
3623
3650
  packet_hash: str,
3624
3651
  profile: dict[str, Any],
3652
+ *,
3653
+ original_round: bool = False,
3625
3654
  ) -> tuple[str, dict[str, Any]]:
3626
3655
  try:
3627
3656
  prior, result_hash = _load_prior_review_result(
3628
- args.completion_review_result_file, 1
3657
+ args.completion_review_result_file, 1, strict_json=original_round
3629
3658
  )
3630
3659
  except GateError as exc:
3631
3660
  raise GateError(exc.reason, "completion_checkpoint_invalid") from exc
@@ -3666,6 +3695,16 @@ def validate_completion_checkpoint(
3666
3695
  and isinstance(prior_gate, dict)
3667
3696
  and prior_gate.get("required") is False
3668
3697
  )
3698
+ findings_checkpoint = (
3699
+ prior.get("status") == "findings"
3700
+ and prior.get("next_action") in (
3701
+ "implementer_self_review", "triage_findings_and_continue_independent_work"
3702
+ )
3703
+ and isinstance(prior_gate, dict)
3704
+ and prior_gate.get("required") is True
3705
+ and isinstance(prior_gate.get("required_triggers"), list)
3706
+ and "findings_returned" in prior_gate["required_triggers"]
3707
+ )
3669
3708
  if (
3670
3709
  prior.get("schema_version") != 3
3671
3710
  or prior.get("mode") not in ("review", "challenge")
@@ -3673,8 +3712,11 @@ def validate_completion_checkpoint(
3673
3712
  prior.get("mode") == "challenge"
3674
3713
  and prior.get("review_chain_tracked") is not True
3675
3714
  )
3676
- or prior.get("status") != "passed"
3677
- or prior.get("findings") != []
3715
+ or not (
3716
+ (prior.get("status") == "passed" and prior.get("findings") == [])
3717
+ or (prior.get("status") == "findings"
3718
+ and isinstance(prior.get("findings"), list) and prior["findings"])
3719
+ )
3678
3720
  or prior.get("candidate_sha256") != packet_hash
3679
3721
  or prior.get("packet_sha256") != packet_hash
3680
3722
  or prior.get("stage") != profile["stage"]
@@ -3728,15 +3770,178 @@ def validate_completion_checkpoint(
3728
3770
  or prior["wording_only_proof_sha256"] is not None
3729
3771
  or "wording_only_scope" not in prior
3730
3772
  or prior["wording_only_scope"] is not None
3731
- or not (final_round_checkpoint or early_challenge_checkpoint)
3773
+ or not (original_round or final_round_checkpoint or early_challenge_checkpoint or findings_checkpoint)
3732
3774
  ):
3733
3775
  raise GateError(
3734
3776
  "completion review result does not bind a passed exact candidate awaiting deep self-review",
3735
3777
  "completion_checkpoint_invalid",
3736
3778
  )
3779
+ if prior["status"] == "findings" and not original_round:
3780
+ raise CompletionFindingsError(
3781
+ "exact-candidate findings require complete source-refutation dispositions"
3782
+ )
3737
3783
  return result_hash, prior
3738
3784
 
3739
3785
 
3786
+ def validate_finding_dispositions(
3787
+ args: argparse.Namespace,
3788
+ packet_hash: str,
3789
+ profile: dict[str, Any],
3790
+ ) -> tuple[str, dict[str, Any], dict[str, Any]]:
3791
+ # These checks bind accountable local source reasoning to immutable receipt
3792
+ # occurrences. They validate structure and provenance, not that reasoning's truth.
3793
+ paths = [*args.prior_review_result_file, args.completion_review_result_file]
3794
+ if not 2 <= len(paths) <= profile["challenge_budget"] + 1:
3795
+ raise CompletionFindingsError("source refutation requires the full review-then-challenge chain")
3796
+ receipt_hashes: list[str] = []
3797
+ focuses: list[str] = []
3798
+ occurrences: list[dict[str, str]] = []
3799
+ seen_occurrences: set[tuple[str, str]] = set()
3800
+ chain_id = None
3801
+ for index, path in enumerate(paths, 1):
3802
+ round_args = argparse.Namespace(**vars(args))
3803
+ round_args.completion_review_result_file = path
3804
+ receipt_hash, receipt = validate_completion_checkpoint(
3805
+ round_args, packet_hash, profile, original_round=True
3806
+ )
3807
+ if index == 1:
3808
+ chain_id = receipt.get("review_chain_id")
3809
+ challenge_index = receipt.get("challenge_index")
3810
+ expected_review_state = (
3811
+ "post_review_budget"
3812
+ if receipt["status"] == "findings" and receipt["autonomous_reviews_remaining"] == 0
3813
+ else "findings_pending"
3814
+ if receipt["status"] == "findings"
3815
+ else "reviewed"
3816
+ )
3817
+ receipt_gate = receipt.get("self_review_gate")
3818
+ if (
3819
+ receipt.get("mode") != ("review" if index == 1 else "challenge")
3820
+ or receipt.get("review_chain_tracked") is not True
3821
+ or receipt.get("review_chain_id") != chain_id
3822
+ or receipt.get("autonomous_review_index") != index
3823
+ or type(challenge_index) is not int
3824
+ or challenge_index != index - 1
3825
+ or receipt.get("challenge_rounds_remaining") != profile["challenge_budget"] - challenge_index
3826
+ or receipt.get("prior_review_result_sha256") != receipt_hashes
3827
+ or receipt.get("prior_challenge_focuses") != focuses
3828
+ or receipt.get("review_state") != expected_review_state
3829
+ or receipt.get("human_decision_required") is not (expected_review_state == "post_review_budget")
3830
+ or (
3831
+ receipt["status"] == "findings"
3832
+ and (
3833
+ not isinstance(receipt_gate, dict)
3834
+ or receipt_gate.get("required") is not True
3835
+ or not isinstance(receipt_gate.get("required_triggers"), list)
3836
+ or "findings_returned" not in receipt_gate["required_triggers"]
3837
+ )
3838
+ )
3839
+ or any(receipt.get(key) is not None for key in (
3840
+ "predecessor_chain_id", "predecessor_result_sha256", "predecessor_candidate_sha256"
3841
+ ))
3842
+ or receipt.get("predecessor_challenge_focuses") not in (None, [])
3843
+ ):
3844
+ raise CompletionFindingsError("source refutation requires original contiguous normal-chain receipts")
3845
+ focus = receipt.get("challenge_focus")
3846
+ if index > 1:
3847
+ if not isinstance(focus, str) or not focus.strip() or focus in focuses:
3848
+ raise CompletionFindingsError("source refutation requires distinct recorded challenge focuses")
3849
+ focuses.append(focus)
3850
+ elif focus is not None:
3851
+ raise CompletionFindingsError("initial review cannot carry a challenge focus")
3852
+ for finding in receipt["findings"]:
3853
+ if (
3854
+ not isinstance(finding, dict)
3855
+ or finding.get("severity") not in ("P0", "P1", "P2")
3856
+ or type(finding.get("line")) is not int
3857
+ or finding["line"] < 1
3858
+ or any(not isinstance(finding.get(key), str) or not finding[key].strip()
3859
+ for key in ("file", "failure_path", "smallest_fix"))
3860
+ ):
3861
+ raise CompletionFindingsError("original receipt contains an invalid finding")
3862
+ try:
3863
+ finding_hash = _canonical_digest(finding)
3864
+ except (UnicodeError, ValueError) as exc:
3865
+ raise CompletionFindingsError("original finding must be canonical UTF-8 JSON") from exc
3866
+ pair = (receipt_hash, finding_hash)
3867
+ if pair in seen_occurrences:
3868
+ # Identical canonical content shares one occurrence identity;
3869
+ # preserve every entry in the original receipt unchanged.
3870
+ continue
3871
+ seen_occurrences.add(pair)
3872
+ occurrences.append({"receipt_sha256": receipt_hash, "finding_sha256": finding_hash})
3873
+ receipt_hashes.append(receipt_hash)
3874
+ if not occurrences:
3875
+ raise CompletionFindingsError("source refutation requires at least one original finding")
3876
+ source = Path(args.finding_dispositions_file)
3877
+ if not source.is_absolute():
3878
+ raise CompletionFindingsError("finding dispositions path must be absolute")
3879
+ try:
3880
+ encoded = read_bounded_regular_file(
3881
+ source,
3882
+ label="finding dispositions",
3883
+ maximum=MAX_RESULT_BYTES,
3884
+ regular_error="finding dispositions must be a bounded regular JSON file",
3885
+ oversized_error="finding dispositions exceed the size limit",
3886
+ reason_code="completion_checkpoint_invalid",
3887
+ )
3888
+ manifest = json.loads(encoded.decode("utf-8"), object_pairs_hook=_unique_json_object)
3889
+ except GateError:
3890
+ raise
3891
+ except (UnicodeError, ValueError) as exc:
3892
+ raise CompletionFindingsError(f"cannot read finding dispositions: {exc}") from exc
3893
+ if (
3894
+ not isinstance(manifest, dict)
3895
+ or set(manifest) != {"schema_version", "candidate_sha256", "review_result_sha256", "dispositions"}
3896
+ or type(manifest.get("schema_version")) is not int
3897
+ or manifest["schema_version"] != 1
3898
+ or manifest.get("candidate_sha256") != packet_hash
3899
+ or manifest.get("review_result_sha256") != receipt_hashes
3900
+ or not isinstance(manifest.get("dispositions"), list)
3901
+ or len(manifest["dispositions"]) != len(occurrences)
3902
+ ):
3903
+ raise CompletionFindingsError("finding dispositions do not bind the exact candidate and full original chain")
3904
+ for disposition, occurrence in zip(manifest["dispositions"], occurrences):
3905
+ if (
3906
+ not isinstance(disposition, dict)
3907
+ or set(disposition) != {"receipt_sha256", "finding_sha256", "disposition", "evidence"}
3908
+ or disposition.get("receipt_sha256") != occurrence["receipt_sha256"]
3909
+ or disposition.get("finding_sha256") != occurrence["finding_sha256"]
3910
+ or disposition.get("disposition") != "source_refuted"
3911
+ ):
3912
+ raise CompletionFindingsError("finding dispositions must cover every original occurrence once in order")
3913
+ evidence = disposition.get("evidence")
3914
+ if (
3915
+ not isinstance(evidence, list)
3916
+ or not evidence
3917
+ or any(
3918
+ not isinstance(item, str) or item != item.strip() or not 1 <= len(item) <= 1000
3919
+ for item in evidence
3920
+ )
3921
+ or len(set(evidence)) != len(evidence)
3922
+ ):
3923
+ raise CompletionFindingsError("each source refutation requires nonempty bounded distinct evidence")
3924
+ if any(
3925
+ ord(char) < 0x20
3926
+ or 0x7F <= ord(char) <= 0x9F
3927
+ or char in "\u2028\u2029"
3928
+ or unicodedata.category(char) == "Cf"
3929
+ for item in evidence
3930
+ for char in item
3931
+ ):
3932
+ raise CompletionFindingsError("source-refutation evidence must not contain control or format characters")
3933
+ try:
3934
+ for item in evidence:
3935
+ item.encode("utf-8")
3936
+ except UnicodeError as exc:
3937
+ raise CompletionFindingsError("source-refutation evidence must be valid UTF-8 text") from exc
3938
+ return receipt_hashes[-1], receipt, {
3939
+ "completion_basis": "source_refuted_findings",
3940
+ "finding_dispositions_sha256": hashlib.sha256(encoded).hexdigest(),
3941
+ "resolved_finding_occurrences": occurrences,
3942
+ }
3943
+
3944
+
3740
3945
  def record_skip(
3741
3946
  result: dict[str, Any],
3742
3947
  client: str,
@@ -3875,6 +4080,7 @@ def build_parser() -> argparse.ArgumentParser:
3875
4080
  parser.add_argument("--prior-review-result-file", action="append", default=[])
3876
4081
  parser.add_argument("--predecessor-chain-result-file", default=None)
3877
4082
  parser.add_argument("--completion-review-result-file")
4083
+ parser.add_argument("--finding-dispositions-file")
3878
4084
  parser.add_argument("--allow-fallback-egress", action="store_true")
3879
4085
  parser.add_argument("--host-remediation-attempted", action="store_true")
3880
4086
  parser.add_argument("--review-harness", action="store_true")
@@ -3991,24 +4197,30 @@ def main(argv: list[str] | None = None) -> int:
3991
4197
  )
3992
4198
  if args.mode == "complete":
3993
4199
  try:
3994
- (
3995
- completion_result_hash,
3996
- completed_review,
3997
- ) = validate_completion_checkpoint(args, packet_hash, profile)
4200
+ if args.finding_dispositions_file:
4201
+ completion_result_hash, completed_review, completion_metadata = validate_finding_dispositions(
4202
+ args, packet_hash, profile
4203
+ )
4204
+ else:
4205
+ completion_result_hash, completed_review = validate_completion_checkpoint(
4206
+ args, packet_hash, profile
4207
+ )
4208
+ completion_metadata = {"completion_basis": "external_pass"}
3998
4209
  except GateError as exc:
4210
+ resolve_findings = isinstance(exc, CompletionFindingsError) or bool(args.finding_dispositions_file)
3999
4211
  result.update(
4000
4212
  reason=exc.reason,
4001
4213
  reason_code=exc.reason_code,
4002
- next_action="run_external_review_for_current_candidate",
4214
+ next_action=("resolve_review_findings" if resolve_findings else "run_external_review_for_current_candidate"),
4003
4215
  review_state="self_reviewing",
4004
4216
  self_review_gate=self_review_gate(
4005
- required_triggers=["material_candidate_change"],
4217
+ required_triggers=(["findings_returned", "before_completion_claim"] if resolve_findings else ["material_candidate_change"]),
4006
4218
  satisfied_triggers=profile["self_review_satisfied_triggers"],
4007
4219
  blocks=["completion_claim"],
4008
4220
  allowed_next_actions=[
4009
4221
  "deep_self_review",
4010
4222
  "continue_implementation",
4011
- "run_external_review_after_self_review",
4223
+ "resolve_review_findings" if resolve_findings else "run_external_review_after_self_review",
4012
4224
  ],
4013
4225
  ),
4014
4226
  )
@@ -4036,6 +4248,7 @@ def main(argv: list[str] | None = None) -> int:
4036
4248
  satisfied_triggers=["before_completion_claim"]
4037
4249
  ),
4038
4250
  )
4251
+ result.update(completion_metadata)
4039
4252
  return emit(result, 0)
4040
4253
  last_reason_code = "no_independent_reviewer_available"
4041
4254
  for client in order:
@@ -4271,6 +4484,7 @@ def main(argv: list[str] | None = None) -> int:
4271
4484
  "post_review_budget_checkpoint"
4272
4485
  )
4273
4486
  allowed_self_review_actions.append("continue_independent_work")
4487
+ allowed_self_review_actions.append("resolve_review_findings")
4274
4488
  current_self_review_gate = self_review_gate(
4275
4489
  required_triggers=required_self_review_triggers,
4276
4490
  satisfied_triggers=profile["self_review_satisfied_triggers"],
@@ -652,7 +652,7 @@ cat >"$WORK/bin/codex" <<'CODEX_STUB'
652
652
  #!/usr/bin/env bash
653
653
  set -u
654
654
  state="$REVIEW_WRAPPER_TEST_STATE"
655
- if [ "${1:-}" = exec ] && [ "${2:-}" = --disable ] && [ "${3:-}" = hooks ] && [ "${4:-}" = --help ]; then
655
+ if [ "${1:-}" = exec ] && [[ " $* " = *" --help "* ]]; then
656
656
  if [ "${STUB_BEHAVIOR:-}" = help_hang ]; then
657
657
  trap '' TERM
658
658
  while :; do /bin/sleep 1; done
@@ -667,16 +667,72 @@ if [ "${1:-}" = exec ] && [ "${2:-}" = --disable ] && [ "${3:-}" = hooks ] && [
667
667
  fi
668
668
  if [ "${1:-}" = features ] && [ "${2:-}" = list ]; then
669
669
  touch "$state/codex_features_invoked"
670
- if [ "${STUB_BEHAVIOR:-pass}" = missing_hooks_feature ]; then
671
- printf '%s\n' 'apply_patch_freeform removed false'
672
- exit 0
673
- fi
674
- if [ "${STUB_BEHAVIOR:-pass}" = removed_hooks_feature ]; then
675
- printf '%s\n' 'hooks removed true'
670
+ hooks_enabled=true
671
+ shell_enabled=true
672
+ while [ "$#" -gt 0 ]; do
673
+ case "$1" in
674
+ --disable)
675
+ case "${2:-}" in hooks) hooks_enabled=false ;; shell_tool) shell_enabled=false ;; esac
676
+ shift 2 ;;
677
+ *) shift ;;
678
+ esac
679
+ done
680
+ case "${STUB_BEHAVIOR:-pass}" in
681
+ missing_hooks_feature) ;;
682
+ removed_hooks_feature) printf '%s\n' 'hooks removed false' ;;
683
+ *) printf 'hooks stable %s\n' "$hooks_enabled" ;;
684
+ esac
685
+ case "${STUB_BEHAVIOR:-pass}" in
686
+ missing_shell_feature) ;;
687
+ removed_shell_feature) printf '%s\n' 'shell_tool removed false' ;;
688
+ ignored_shell_disable) printf '%s\n' 'shell_tool stable true' ;;
689
+ *) printf 'shell_tool stable %s\n' "$shell_enabled" ;;
690
+ esac
691
+ exit 0
692
+ fi
693
+ if [ "${1:-}" = mcp ] && [ "${2:-}" = list ]; then
694
+ touch "$state/codex_mcp_probe_invoked"
695
+ if [ "${STUB_BEHAVIOR:-pass}" = mcp_capability_missing ]; then
696
+ printf '%s\n' '[]'
676
697
  exit 0
677
698
  fi
678
- printf '%s\n' 'hooks stable true'
679
- exit 0
699
+ python3 - "$@" <<'PY_MCP_LIST'
700
+ import json, os, sys, tomllib
701
+
702
+ arguments = iter(sys.argv[1:])
703
+ servers = {}
704
+ inherited_names = {
705
+ "inherited_mcp": "unrelated",
706
+ "inherited_mcp_dot": "unrelated.name",
707
+ "inherited_mcp_space": "unrelated name",
708
+ "inherited_mcp_quote": 'unrelated"name',
709
+ }
710
+ inherited_name = inherited_names.get(os.environ.get("STUB_BEHAVIOR"))
711
+ if inherited_name is not None:
712
+ servers[inherited_name] = {"command": "/bin/false", "args": [], "enabled": True}
713
+ for argument in arguments:
714
+ if argument in {"-c", "--config"}:
715
+ # Codex splits the override path separately from its TOML value;
716
+ # quotes in a dotted left-hand key are not TOML key quoting.
717
+ key, value = next(arguments).split("=", 1)
718
+ decoded = tomllib.loads("value=" + value)["value"]
719
+ if key == "mcp_servers":
720
+ configured_servers = decoded
721
+ elif key.startswith("mcp_servers."):
722
+ _, name, field = key.split(".")
723
+ configured_servers = {name: {field: decoded}}
724
+ else:
725
+ continue
726
+ for name, settings in configured_servers.items():
727
+ servers.setdefault(name, {}).update(settings)
728
+ print(json.dumps([
729
+ {"name": name, "enabled": server.get("enabled", True),
730
+ "transport": {"type": "stdio", "command": server["command"],
731
+ "args": server.get("args", [])}}
732
+ for name, server in servers.items()
733
+ ]))
734
+ PY_MCP_LIST
735
+ exit $?
680
736
  fi
681
737
  touch "$state/codex_invoked"
682
738
  printf '%s' "$0" >"$state/codex_argv0"
@@ -687,14 +743,19 @@ has_read_only=no
687
743
  has_ephemeral=no
688
744
  has_ignore_rules=no
689
745
  has_hooks_disabled=no
746
+ has_shell_disabled=no
690
747
  workspace=""
748
+ : >"$state/codex_configs"
691
749
  while [ "$#" -gt 0 ]; do
692
750
  case "$1" in
693
751
  --output-last-message) last_message="$2"; shift 2 ;;
694
752
  --model|-m) has_model=yes; shift 2 ;;
695
753
  --sandbox) [ "$2" = read-only ] && has_read_only=yes; shift 2 ;;
696
754
  --ephemeral) has_ephemeral=yes; shift ;;
697
- --disable) [ "$2" = hooks ] && has_hooks_disabled=yes; shift 2 ;;
755
+ --disable)
756
+ case "$2" in hooks) has_hooks_disabled=yes ;; shell_tool) has_shell_disabled=yes ;; esac
757
+ shift 2 ;;
758
+ -c|--config) printf '%s\n' "$2" >>"$state/codex_configs"; shift 2 ;;
698
759
  --ignore-rules) has_ignore_rules=yes; shift ;;
699
760
  -C) workspace="$2"; shift 2 ;;
700
761
  *) shift ;;
@@ -705,6 +766,7 @@ printf '%s' "$has_read_only" >"$state/codex_read_only"
705
766
  printf '%s' "$has_ephemeral" >"$state/codex_ephemeral"
706
767
  printf '%s' "$has_ignore_rules" >"$state/codex_ignore_rules"
707
768
  printf '%s' "$has_hooks_disabled" >"$state/codex_hooks_disabled"
769
+ printf '%s' "$has_shell_disabled" >"$state/codex_shell_disabled"
708
770
  [ -z "$workspace" ] || printf '%s' "$workspace" >"$state/codex_workspace"
709
771
  if [ -n "$workspace" ] && [ -L "$workspace/.agents/skills/testing-strategy" ]; then
710
772
  readlink "$workspace/.agents/skills/testing-strategy" >"$state/codex_skill_link"
@@ -741,7 +803,8 @@ fi
741
803
  printf '%s\n' '{"type":"thread.started","thread_id":"test-thread"}'
742
804
  printf '%s\n' '{"type":"turn.started"}'
743
805
  case "$behavior" in
744
- pass|skills_budget_warning|skills_budget_warning_after_concern|hook_trust_warning|hook_trust_warning_started|hook_trust_warning_repeated_after_concern|unknown_error_valid_result) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
806
+ packet_read|packet_search|packet_tampered) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
807
+ pass|shell_disable_required|missing_shell_feature|removed_shell_feature|ignored_shell_disable|mcp_capability_missing|inherited_mcp|inherited_mcp_dot|inherited_mcp_space|inherited_mcp_quote|skills_budget_warning|skills_budget_warning_after_concern|hook_trust_warning|hook_trust_warning_started|hook_trust_warning_repeated_after_concern|unknown_error_valid_result) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
745
808
  legacy_pass) printf '%s\n' '{"status":"passed","findings":[]}' >"$last_message" ;;
746
809
  stream_gap) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
747
810
  tool) printf '%s\n' '{"status":"passed","concern_results":[{"concern":"correctness","conclusion":"Independently checked correctness against the frozen candidate."}],"findings":[]}' >"$last_message" ;;
@@ -750,9 +813,46 @@ case "$behavior" in
750
813
  invalid) printf '%s\n' 'not-json' >"$last_message" ;;
751
814
  invalid_concern) printf '%s\n' 'P1 src/example.py:7 concern without valid JSON' >"$last_message" ;;
752
815
  esac
816
+ if [[ "$behavior" = packet_* ]]; then
817
+ python3 - "$state" "$behavior" <<'PY_PACKET_CALL'
818
+ import json, subprocess, sys, tomllib
819
+ from pathlib import Path
820
+
821
+ state, behavior = Path(sys.argv[1]), sys.argv[2]
822
+ servers = {}
823
+ for override in (state / "codex_configs").read_text().splitlines():
824
+ key, value = override.split("=", 1)
825
+ if key == "mcp_servers":
826
+ servers = tomllib.loads("servers=" + value)["servers"]
827
+ server = servers["code_review_packet"]
828
+ tool = "search_packet" if behavior == "packet_search" else "read_packet"
829
+ arguments = ({"query": "diff --git", "byte_offset": 0, "limit": 1}
830
+ if tool == "search_packet" else {"byte_offset": 0, "max_bytes": 46000})
831
+ request = {"jsonrpc": "2.0", "id": 1, "method": "tools/call",
832
+ "params": {"name": tool, "arguments": arguments}}
833
+ response = subprocess.run([server["command"], *server["args"]],
834
+ input=json.dumps(request) + "\n", text=True,
835
+ capture_output=True, timeout=5, check=True)
836
+ result = json.loads(response.stdout)["result"]
837
+ assert not result.get("isError"), result
838
+ if behavior == "packet_tampered":
839
+ result["content"][0]["text"] += "forged packet bytes"
840
+ item = {"type": "mcp_tool_call", "id": "packet-call-1",
841
+ "server": "code_review_packet", "tool": tool, "arguments": arguments,
842
+ "status": "in_progress"}
843
+ print(json.dumps({"type": "item.started", "item": item}))
844
+ item.update(status="completed", result=result)
845
+ print(json.dumps({"type": "item.completed", "item": item}))
846
+ (state / "codex_packet_call").write_text(tool)
847
+ PY_PACKET_CALL
848
+ [ "$?" = 0 ] || exit 42
849
+ fi
753
850
  if [ "$behavior" = stream_gap ]; then
754
851
  printf '%s\n' '{"type":"item.completed","item":{"type":"error","message":"in-process app-server event stream lagged; dropped 2 events"}}'
755
852
  fi
853
+ if [ "$behavior" = shell_disable_required ] && [ "$has_shell_disabled" != yes ]; then
854
+ printf '%s\n' '{"type":"item.completed","item":{"type":"command_execution","command":"pwd"}}'
855
+ fi
756
856
  if [ "$behavior" = tool ]; then
757
857
  printf '%s\n' '{"type":"item.completed","item":{"type":"command_execution","command":"pwd"}}'
758
858
  elif [ "$behavior" = foreign_tool ]; then
@@ -1676,6 +1776,60 @@ out="$(run_codex removed_hooks_feature)"; rc=$?
1676
1776
  check "Codex with a removed hooks feature key falls back before inference" \
1677
1777
  '[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_hook_disable_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
1678
1778
 
1779
+ out="$(run_codex shell_disable_required)"; rc=$?
1780
+ check "Codex disables shell execution before the reviewer can request a command" \
1781
+ '[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(cat "$WORK/state/codex_shell_disabled")" = yes ]'
1782
+
1783
+ for shell_capability in missing_shell_feature removed_shell_feature ignored_shell_disable; do
1784
+ rm -f "$WORK/state/codex_invoked"
1785
+ out="$(run_codex "$shell_capability")"; rc=$?
1786
+ check "Codex $shell_capability falls back before inference" \
1787
+ '[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_shell_disable_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
1788
+ done
1789
+
1790
+ rm -f "$WORK/state/codex_invoked"
1791
+ out="$(run_codex mcp_capability_missing)"; rc=$?
1792
+ check "Codex without the bounded packet MCP server falls back before inference" \
1793
+ '[ "$rc" = 2 ] && [ "$(field reason "$out")" = codex_packet_tools_unavailable ] && [ "$(field reason_code "$out")" = capability_missing ] && [ "$(field cascade_eligible "$out")" = True ] && [ ! -e "$WORK/state/codex_invoked" ]'
1794
+
1795
+ for packet_tool in read search; do
1796
+ rm -f "$WORK/state/codex_packet_call"
1797
+ out="$(run_codex "packet_$packet_tool")"; rc=$?
1798
+ check "Codex accepts a real frozen packet $packet_tool through wrapper and parser" \
1799
+ '[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ "$(cat "$WORK/state/codex_packet_call")" = "${packet_tool}_packet" ]'
1800
+ done
1801
+ out="$(run_codex packet_tampered)"; rc=$?
1802
+ check "Codex rejects altered packet tool bytes through wrapper and parser" \
1803
+ '[ "$rc" = 2 ] && [ "$(field reason_code "$out")" = binding_mismatch ] && [ "$(field cascade_eligible "$out")" = False ]'
1804
+
1805
+ for inherited_mcp_case in inherited_mcp inherited_mcp_dot inherited_mcp_space inherited_mcp_quote; do
1806
+ case "$inherited_mcp_case" in
1807
+ inherited_mcp) inherited_mcp_name=unrelated ;;
1808
+ inherited_mcp_dot) inherited_mcp_name=unrelated.name ;;
1809
+ inherited_mcp_space) inherited_mcp_name='unrelated name' ;;
1810
+ inherited_mcp_quote) inherited_mcp_name='unrelated"name' ;;
1811
+ esac
1812
+ rm -f "$WORK/state/codex_invoked" "$WORK/state/codex_configs"
1813
+ out="$(run_codex "$inherited_mcp_case")"; rc=$?
1814
+ inherited_mcp_disabled="$(python3 - "$inherited_mcp_name" "$WORK/state/codex_configs" <<'PY_MCP_DISABLED'
1815
+ import sys, tomllib
1816
+ from pathlib import Path
1817
+
1818
+ path = Path(sys.argv[2])
1819
+ tables = []
1820
+ for override in path.read_text().splitlines() if path.exists() else []:
1821
+ key, value = override.split("=", 1)
1822
+ if key.strip() == "mcp_servers":
1823
+ tables.append(tomllib.loads("servers=" + value)["servers"])
1824
+ print(len(tables) == 1
1825
+ and tables[0].get(sys.argv[1], {}).get("enabled") is False
1826
+ and "code_review_packet" in tables[0])
1827
+ PY_MCP_DISABLED
1828
+ )"
1829
+ check "Codex encodes and disables the inherited MCP key ($inherited_mcp_case)" \
1830
+ '[ "$rc" = 0 ] && [ "$(field status "$out")" = passed ] && [ -e "$WORK/state/codex_invoked" ] && [ "$inherited_mcp_disabled" = True ]'
1831
+ done
1832
+
1679
1833
  rm -f "$WORK/state/codex_invoked" "$WORK/state/codex_help_invoked"
1680
1834
  probe_started=$SECONDS
1681
1835
  out="$(run_codex help_hang)"; rc=$?