@ccoalm/ccl-skills 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +1 -1
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +39 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
  35. package/dist/assets/release.json +41 -36
  36. package/package.json +1 -1
@@ -334,11 +334,13 @@ def parse_json_object(text: str) -> object | None:
334
334
  # block may occur. Ground truth comes from the stream, not the model's reply
335
335
  # token, which can hallucinate TOOL_ENABLED even when no tool exists.
336
336
  PROBE_TOOL = "bash"
337
+ # The only customization surface still required to be EMPTY. An inherited MCP
338
+ # server is invocable capability, so it is a breach. Skill, command and plugin
339
+ # lists are host/plugin VOCABULARY: nothing in them is invocable while `tools`
340
+ # is pinned to the expected set and the tool_use scan is clean, so they are
341
+ # recorded metadata and never judged -- see KNOWN_VOCABULARY_INIT_FIELDS.
337
342
  REQUIRED_EMPTY_INIT_FIELDS = (
338
343
  "mcp_servers",
339
- "slash_commands",
340
- "skills",
341
- "plugins",
342
344
  )
343
345
  # Fields that must be PRESENT (not merely empty). `permissionMode` is the only
344
346
  # report of the reviewer's authority, and authority is scalar-shaped — which is
@@ -444,90 +446,21 @@ KNOWN_SAFE_INIT_METADATA_FIELDS = {
444
446
  "uuid",
445
447
  "fast_mode_state",
446
448
  }
447
- KNOWN_SAFE_INIT_LIST_METADATA_FIELDS = {
448
- # Claude Code 2.1.233 added this classification-only subset of the already
449
- # declared slash_commands surface. It is safe only when every entry is a
450
- # whole plain string already present in slash_commands; otherwise the new
451
- # schema remains unverifiable and the lane fails closed.
449
+ # Host/plugin vocabulary the init event enumerates. Deliberately NOT a
450
+ # boundary and never judged by name, shape, or origin: an entry here becomes
451
+ # invocable only through a tool, and the exact `tools` allowlist plus the
452
+ # tool_use scan already pin that. Judging these lists against a snapshot of the
453
+ # host's own built-in names turned every CLI release that shipped a new skill or
454
+ # command into a reviewer-lane outage while proving nothing about isolation, and
455
+ # it made an uninstalled or older plugin a refusal instead of a review without
456
+ # owner skills.
457
+ KNOWN_VOCABULARY_INIT_FIELDS = (
458
+ "slash_commands",
452
459
  "terminal_slash_commands",
453
- }
460
+ "skills",
461
+ "plugins",
462
+ )
454
463
  KNOWN_SAFE_PERMISSION_MODES = {"default", "plan"}
455
- KNOWN_SAFE_BUILTIN_SKILLS = {
456
- "batch",
457
- "claude-api",
458
- "code-review",
459
- "dataviz",
460
- "debug",
461
- "deep-research",
462
- "design-sync",
463
- "doctor",
464
- "fewer-permission-prompts",
465
- "loop",
466
- "run",
467
- "run-skill-generator",
468
- "schedule",
469
- "simplify",
470
- "update-config",
471
- "verify",
472
- }
473
- KNOWN_SAFE_BUILTIN_SLASH_COMMANDS = {
474
- "__remote-workflow",
475
- "agents",
476
- "batch",
477
- "claude-api",
478
- "clear",
479
- "code-review",
480
- "color",
481
- "compact",
482
- "config",
483
- "context",
484
- "dataviz",
485
- "debug",
486
- "deep-research",
487
- "design",
488
- "design-consent",
489
- "design-revoke",
490
- "design-sync",
491
- "doctor",
492
- "effort",
493
- "extra-usage",
494
- "fast",
495
- "fewer-permission-prompts",
496
- "goal",
497
- "heapdump",
498
- # `/import` arrived in a CLI release after this snapshot was written, which
499
- # is how the snapshot's cost was discovered: an unrecognised name used to be
500
- # classified as a proven customization -- terminal and non-cascadable -- so
501
- # one new built-in took out the whole reviewer lane. That is fixed at the
502
- # CLASS level now (see is_bare_host_identifier): a bare name this list does
503
- # not know is unverifiable and cascades. This list therefore no longer
504
- # decides whether review is POSSIBLE, only whether the Claude lane is the
505
- # one that serves it, so a missing name costs a client switch instead of the
506
- # capability. Keep it accurate where it is cheap; do not treat adding a name
507
- # as the fix for a lane outage.
508
- "import",
509
- "init",
510
- "insights",
511
- "loop",
512
- "mcp",
513
- "model",
514
- "recap",
515
- "reload-skills",
516
- "rename",
517
- "review",
518
- "run",
519
- "run-skill-generator",
520
- "schedule",
521
- "security-review",
522
- "simplify",
523
- "team-onboarding",
524
- "ultrareview",
525
- "update-config",
526
- "usage",
527
- "usage-credits",
528
- "verify",
529
- "workflow-launch-exec",
530
- }
531
464
 
532
465
 
533
466
  def stream_events(text: str) -> list[dict]:
@@ -644,22 +577,16 @@ def init_declared_tools(events: list[dict]) -> set[str] | None:
644
577
 
645
578
  def init_surface_state(
646
579
  events: list[dict],
647
- expected_tools: set[str] | None = None,
648
- expected_native_skills: set[str] | None = None,
649
- required_native_skills: set[str] | None = None,
650
- host_init_baseline: dict[str, object] | None = None,
651
- ) -> tuple[
652
- set[str], set[str], set[str], set[str], set[str], set[str], set[str]
653
- ] | None:
580
+ ) -> tuple[set[str], set[str], set[str], set[str], set[str], set[str]] | None:
654
581
  """Return init isolation state across all init events.
655
582
 
656
- Result fields: missing/invalid required fields, non-empty customization
583
+ Result fields: missing/invalid required fields, non-empty capability
657
584
  fields, declared tool names, unrecognized surface-shaped fields, fields
658
- declaring an unsafe value, fields whose authority is unverifiable, and
659
- unclassifiable host-vocabulary identifiers. A missing or wrongly typed
660
- required field is not treated as empty, because that would turn a CLI schema
661
- change into a silent gate bypass. Unknown *scalar* metadata is tolerated on
662
- purpose — see the drift policy on KNOWN_SAFE_INIT_METADATA_FIELDS.
585
+ declaring an unsafe value, and fields whose authority is unverifiable. A
586
+ missing or wrongly typed required field is not treated as empty, because
587
+ that would turn a CLI schema change into a silent gate bypass. Unknown
588
+ *scalar* metadata is tolerated on purpose — see the drift policy on
589
+ KNOWN_SAFE_INIT_METADATA_FIELDS.
663
590
 
664
591
  Unsafe *values* are kept apart from unrecognized *shapes* on purpose: an
665
592
  unknown schema addition means this lane cannot verify isolation (report it,
@@ -668,31 +595,11 @@ def init_surface_state(
668
595
  must terminate the lane. Merging them would let a real privilege escalation
669
596
  inherit the drift class's softer next_action.
670
597
 
671
- Unrecognized *identifiers* split the same way, and for the same reason: a
672
- bare command/skill name outside the built-in snapshot cannot be shown to be
673
- a customization, because the snapshot is of a vocabulary the host owns, so
674
- it reports as unverifiable. A namespaced, path-shaped, duplicated or
675
- unparseable entry IS a customization and stays in `nonempty`.
598
+ Skill, command and plugin lists are vocabulary, not capability: they are
599
+ read for nothing and may hold any value. Isolation is proven by the exact
600
+ `tools` allowlist, the tool_use scan, the empty MCP list and the pinned
601
+ `permissionMode` — none of which any vocabulary entry can affect.
676
602
  """
677
- expected_native_skills = expected_native_skills or set()
678
- required_native_skills = {
679
- name.lower() for name in (required_native_skills or set())
680
- }
681
- baseline_commands = (
682
- set(host_init_baseline.get("slash_commands", ()))
683
- if host_init_baseline
684
- else set()
685
- )
686
- baseline_skills = (
687
- set(host_init_baseline.get("skills", ()))
688
- if host_init_baseline
689
- else set()
690
- )
691
- baseline_version = (
692
- host_init_baseline.get("claude_code_version")
693
- if host_init_baseline
694
- else None
695
- )
696
603
  saw_init = False
697
604
  missing_or_invalid: set[str] = set()
698
605
  nonempty: set[str] = set()
@@ -706,23 +613,12 @@ def init_surface_state(
706
613
  # change exists to remove. A KNOWN field carrying a KNOWN-unsafe value stays
707
614
  # terminal in `unsafe_values`, because that one is proven.
708
615
  unverifiable_authority: set[str] = set()
709
- # Host vocabulary this snapshot does not recognise. Kept apart from
710
- # `nonempty` for the same reason `unverifiable_authority` is: reporting an
711
- # unverifiable thing in the proven-breach class is what turned a routine CLI
712
- # release into a total review outage. Bare identifiers only — see
713
- # is_bare_host_identifier for what stays terminal.
714
- unclassifiable_vocabulary: set[str] = set()
715
- declared_customizations: dict[str, set[str]] = {
716
- "slash_commands": set(),
717
- "skills": set(),
718
- "plugins": set(),
719
- }
720
616
  known_fields = {
721
617
  "tools",
722
618
  "capabilities",
723
619
  *REQUIRED_EMPTY_INIT_FIELDS,
620
+ *KNOWN_VOCABULARY_INIT_FIELDS,
724
621
  *KNOWN_SAFE_INIT_METADATA_FIELDS,
725
- *KNOWN_SAFE_INIT_LIST_METADATA_FIELDS,
726
622
  }
727
623
  for ev in events:
728
624
  if ev.get("type") != "system" or ev.get("subtype") != "init":
@@ -744,49 +640,6 @@ def init_surface_state(
744
640
  value = ev.get(field)
745
641
  if not isinstance(value, list):
746
642
  missing_or_invalid.add(field)
747
- elif field in declared_customizations:
748
- identifiers = [
749
- customization_entry_identifier(entry) for entry in value
750
- ]
751
- declared_customizations[field].update(identifiers)
752
- if len(identifiers) != len(set(identifiers)):
753
- nonempty.add(field)
754
- if value and expected_native_skills:
755
- for entry in value:
756
- identifier = customization_entry_identifier(entry)
757
- if field in HOST_VOCABULARY_FIELDS and (
758
- not host_entry_is_whole(entry, identifier)
759
- ):
760
- # Nothing may be discarded to reach a verdict on
761
- # these fields — see host_entry_is_whole. Runs BEFORE
762
- # the allowlist, because the allowlist consumes the
763
- # same lossy identifier and would clear the entry
764
- # outright on a truncated or dict-supplied name.
765
- # Legitimate namespaced entries are unaffected: their
766
- # whole value IS the identifier.
767
- nonempty.add(field)
768
- continue
769
- if customization_entry_allowed(
770
- field,
771
- entry,
772
- expected_native_skills,
773
- baseline_commands,
774
- baseline_skills,
775
- ):
776
- continue
777
- if field in HOST_VOCABULARY_FIELDS and (
778
- is_bare_host_identifier(identifier)
779
- ):
780
- # Cannot be shown to be a customization: the
781
- # allowlist it failed is a snapshot of a vocabulary
782
- # the host owns. Refuse, but stay cascadable.
783
- unclassifiable_vocabulary.add(
784
- f"{field}:{identifier}"
785
- )
786
- else:
787
- nonempty.add(field)
788
- elif value:
789
- nonempty.add(field)
790
643
  elif value:
791
644
  nonempty.add(field)
792
645
  # Per EVENT, not on the union across events: a later init that drops or
@@ -824,73 +677,17 @@ def init_surface_state(
824
677
  elif field == "permissionMode":
825
678
  if value not in KNOWN_SAFE_PERMISSION_MODES:
826
679
  unsafe_values.add(field)
827
- elif field in KNOWN_SAFE_INIT_LIST_METADATA_FIELDS:
828
- if not isinstance(value, list) or any(
829
- not isinstance(item, str) for item in value
830
- ):
831
- unknown_fields.add(field)
832
- else:
833
- identifiers = [
834
- customization_entry_identifier(item) for item in value
835
- ]
836
- if (
837
- len(identifiers) != len(set(identifiers))
838
- or any(
839
- not host_entry_is_whole(item, identifier)
840
- for item, identifier in zip(value, identifiers)
841
- )
842
- or not set(identifiers).issubset(
843
- declared_customizations["slash_commands"]
844
- )
845
- ):
846
- unknown_fields.add(field)
680
+ elif field in KNOWN_VOCABULARY_INIT_FIELDS:
681
+ # Vocabulary, not capability: any value, any shape. Whatever a
682
+ # CLI release or an installed plugin lists here cannot be
683
+ # invoked past the pinned `tools` set.
684
+ continue
847
685
  elif field in KNOWN_SAFE_INIT_METADATA_FIELDS and isinstance(
848
686
  value, (list, dict)
849
687
  ):
850
688
  unknown_fields.add(field)
851
- if baseline_version is not None and ev.get("claude_code_version") != baseline_version:
852
- # The two invocations no longer prove one same-version host
853
- # vocabulary snapshot. Refuse this lane, but treat the mismatch as
854
- # capability drift rather than a proven tool/authority breach so a
855
- # different reviewer may continue.
856
- unknown_fields.add("claude_code_version:host-baseline-mismatch")
857
689
  if not saw_init:
858
690
  return None
859
- if expected_native_skills:
860
- if "ccl-skills" not in declared_customizations["plugins"]:
861
- missing_or_invalid.add("plugins:ccl-skills")
862
- ambiguous_required_skills = (
863
- required_native_skills & KNOWN_SAFE_BUILTIN_SKILLS
864
- )
865
- for skill_name in ambiguous_required_skills:
866
- missing_or_invalid.add(f"skills:ambiguous-selected-owner:{skill_name}")
867
- expected_ccl_skill_identifiers = {
868
- identifier
869
- for skill_name in expected_native_skills
870
- for identifier in (skill_name, f"ccl-skills:{skill_name}")
871
- }
872
- declared_ccl_skills = (
873
- declared_customizations["skills"]
874
- & expected_ccl_skill_identifiers
875
- ) - KNOWN_SAFE_BUILTIN_SKILLS
876
- if declared_ccl_skills:
877
- for skill_name in required_native_skills:
878
- namespaced = f"ccl-skills:{skill_name}"
879
- if not {
880
- skill_name,
881
- namespaced,
882
- } & declared_customizations["skills"]:
883
- missing_or_invalid.add(f"skills:{skill_name}")
884
- declared_ccl_commands = {
885
- identifier
886
- for identifier in declared_customizations["slash_commands"]
887
- if identifier.startswith("ccl-skills:")
888
- }
889
- if declared_ccl_commands:
890
- for skill_name in required_native_skills:
891
- namespaced = f"ccl-skills:{skill_name}"
892
- if namespaced not in declared_customizations["slash_commands"]:
893
- missing_or_invalid.add(f"slash_commands:{namespaced}")
894
691
  return (
895
692
  missing_or_invalid,
896
693
  nonempty,
@@ -898,286 +695,7 @@ def init_surface_state(
898
695
  unknown_fields,
899
696
  unsafe_values,
900
697
  unverifiable_authority,
901
- unclassifiable_vocabulary,
902
- )
903
-
904
-
905
- def customization_entry_identifier(entry: object) -> str:
906
- raw_identifier = ""
907
- if isinstance(entry, str):
908
- raw_identifier = entry.strip().split(maxsplit=1)[0]
909
- elif isinstance(entry, dict):
910
- for key in ("name", "command", "id"):
911
- candidate = entry.get(key)
912
- if isinstance(candidate, str) and candidate.strip():
913
- raw_identifier = candidate.strip().split(maxsplit=1)[0]
914
- break
915
- if not re.fullmatch(r"[A-Za-z0-9_./:@+-]{1,120}", raw_identifier):
916
- return "<unidentified>"
917
- return raw_identifier.lower().lstrip("/")
918
-
919
-
920
- def customization_entry_allowed(
921
- field: str,
922
- entry: object,
923
- expected_native_skills: set[str],
924
- baseline_commands: set[str] | None = None,
925
- baseline_skills: set[str] | None = None,
926
- ) -> bool:
927
- identifier = customization_entry_identifier(entry)
928
- baseline_commands = baseline_commands or set()
929
- # Baseline skills remain available to drift diagnostics, but never grant
930
- # formal-run authority: a future safe-mode regression could otherwise turn
931
- # one leaked user skill into an accepted callable capability.
932
- baseline_skills = baseline_skills or set()
933
- selected_names = {name.lower() for name in expected_native_skills}
934
- selected_namespaced = {
935
- f"ccl-skills:{name}" for name in selected_names
936
- }
937
- if field == "plugins":
938
- return identifier == "ccl-skills"
939
- if field == "skills":
940
- return (
941
- identifier in KNOWN_SAFE_BUILTIN_SKILLS
942
- or identifier in selected_names
943
- or identifier in selected_namespaced
944
- )
945
- if field == "slash_commands":
946
- return (
947
- identifier in KNOWN_SAFE_BUILTIN_SLASH_COMMANDS
948
- or identifier in baseline_commands
949
- or identifier in selected_namespaced
950
- )
951
- return False
952
-
953
-
954
- # Fields whose entries, in a real run, come from the HOST's built-in vocabulary
955
- # rather than from anything this repo installs: measured at CLI 2.1.220, the
956
- # review-skill invocation reports 46 built-in commands and 16 built-in skills and
957
- # nothing else. `plugins` is deliberately absent -- a plugin is user-installed by
958
- # definition, and the expected set there is exactly the one this repo owns.
959
- HOST_VOCABULARY_FIELDS = ("slash_commands", "skills")
960
- # A bare identifier: no namespace, no path. Deliberately NOT "anything the
961
- # allowlist missed" -- the discriminator has to be a property of the identifier
962
- # itself, because the allowlist is the thing that cannot be trusted to be
963
- # current.
964
- BARE_HOST_IDENTIFIER = re.compile(r"[a-z0-9][a-z0-9_.-]*\Z")
965
-
966
-
967
- def host_entry_is_whole(entry: object, identifier: str) -> bool:
968
- """True when nothing about a host-vocabulary entry was discarded to read it.
969
-
970
- `customization_entry_identifier` is deliberately lossy — it keeps the first
971
- whitespace-delimited token and reads a dict's `name` — which is right for a
972
- DIAGNOSTIC string and wrong for a trust decision. Three review findings in
973
- this slice were one shape: evidence present in the entry but never read. A
974
- dict hid a path in a sibling key; a dict hid it under an allowed built-in
975
- name; and `"brand-new evil-plugin:pwn"` truncated to `brand-new` while the
976
- discarded suffix was the very proof of a customization.
977
-
978
- So the classification consumes the WHOLE value: the entry must be a plain
979
- string whose normalization — case folding and at most one leading `/`, the
980
- only normalization the identifier helper itself applies — reproduces the
981
- identifier exactly. Any remainder, any other shape, and the entry is a
982
- customization this parser cannot clear. Checked BEFORE the allowlist,
983
- because the allowlist reads the same truncated token.
984
- """
985
- if not isinstance(entry, str):
986
- return False
987
- # Deliberately NO `.strip()`. The first version of this function called it,
988
- # which reproduced inside the fix the exact lossiness the fix exists to
989
- # reject: `"import "` stripped to `import`, compared equal to the identifier,
990
- # was declared whole, and the allowlist then ACCEPTED it with isolation
991
- # reported verified. Surrounding whitespace is part of the value, so an entry
992
- # carrying any is not a plain host name.
993
- normalized = entry.lower()
994
- if normalized.startswith("/"):
995
- normalized = normalized[1:]
996
- return normalized == identifier
997
-
998
-
999
- def is_bare_host_identifier(identifier: str) -> bool:
1000
- """True when a disallowed entry cannot be PROVEN to be a customization.
1001
-
1002
- Only ever consulted for a PLAIN STRING entry — see the caller. A dict entry
1003
- is a structured descriptor whose other keys this parser does not validate,
1004
- so `{"name": "brand-new", "command": "/x/y"}` would otherwise be classified
1005
- on its bare `name` while a sibling key carried path-shaped proof of a real
1006
- customization. That is not the unclassifiable case this class is for: the
1007
- evidence was present and merely unread. Measured against the real CLI, both
1008
- host-vocabulary fields arrive as plain strings (only `plugins`, already
1009
- excluded, uses dicts), so refusing dicts here costs nothing operationally.
1010
-
1011
- `customization_entry_identifier` has already lowercased, stripped a leading
1012
- `/`, and replaced anything it could not parse with `<unidentified>`. What is
1013
- left is bare only if it carries no namespace separator and no path
1014
- separator, so each of these stays a proven breach:
1015
-
1016
- * `evil-plugin:pwn` -- a namespace proves a surface beyond the one
1017
- expected plugin, which is a customization, not host vocabulary.
1018
- * `dir/cmd` -- no host built-in is spelled as a path.
1019
- * `<unidentified>` -- an entry that failed the charset or was not a
1020
- string/dict is not evidence of anything, least of all of host origin.
1021
-
1022
- Note what this does NOT claim: a bare name may still be a user-authored
1023
- command. The point is that we cannot tell, and the unclassifiable case
1024
- belongs in the class that refuses and cascades rather than the class that
1025
- refuses and destroys the capability. Isolation itself is still proven only
1026
- by the exact `tools` allowlist, the tool_use scan, and the value-pinned
1027
- `permissionMode` -- none of which any identifier here can affect.
1028
- """
1029
- return bool(BARE_HOST_IDENTIFIER.fullmatch(identifier))
1030
-
1031
-
1032
- def load_host_init_baseline(path: str) -> tuple[dict[str, object] | None, str | None]:
1033
- """Read one safe-mode, no-plugin init capture as version-owned vocabulary.
1034
-
1035
- The baseline is not a verdict and its model result is ignored. It may name
1036
- host commands and skills only because the wrapper invocation disables
1037
- setting sources, CLAUDE.md, auto-memory, MCP, tools, and plugins. Any tool,
1038
- plugin, malformed identifier, authority drift, or unknown non-empty surface
1039
- keeps the baseline unusable rather than laundering it into the main run.
1040
- """
1041
- try:
1042
- text = Path(path).read_text(encoding="utf-8", errors="replace")
1043
- except OSError:
1044
- return None, "Claude host-vocabulary baseline could not be read"
1045
- events = stream_events(normalize(text))
1046
- stream_like, stream_corrupt = stream_classification(text, events)
1047
- init_events = [
1048
- event
1049
- for event in events
1050
- if event.get("type") == "system" and event.get("subtype") == "init"
1051
- ]
1052
- if not stream_like or stream_corrupt or len(init_events) != 1:
1053
- return None, "Claude host-vocabulary baseline lacks one intact init event"
1054
- if invoked_tool_names(events):
1055
- return None, "Claude host-vocabulary baseline invoked a tool"
1056
- init_event = init_events[0]
1057
- if init_event.get("tools") != []:
1058
- return None, "Claude host-vocabulary baseline exposed a tool"
1059
- # Keep this set-derived rather than spelling today's MCP/plugin fields
1060
- # twice. If the formal init adds another required-empty customization
1061
- # surface, the baseline must prove it empty before its command vocabulary
1062
- # can carry authority. Known metadata such as `agents` is deliberately not
1063
- # in REQUIRED_EMPTY_INIT_FIELDS: with tools pinned empty it is descriptive,
1064
- # not model-invocable.
1065
- for field in REQUIRED_EMPTY_INIT_FIELDS:
1066
- if field not in HOST_VOCABULARY_FIELDS and init_event.get(field) != []:
1067
- return None, f"Claude host-vocabulary baseline exposed {field}"
1068
- if init_event.get("permissionMode") != "default":
1069
- return None, "Claude host-vocabulary baseline changed permission authority"
1070
- version = init_event.get("claude_code_version")
1071
- if not isinstance(version, str) or not version or len(version) > 120:
1072
- return None, "Claude host-vocabulary baseline lacks a bounded CLI version"
1073
-
1074
- surfaces: dict[str, tuple[str, ...]] = {}
1075
- for field in HOST_VOCABULARY_FIELDS:
1076
- known_host_identifiers = (
1077
- KNOWN_SAFE_BUILTIN_SLASH_COMMANDS
1078
- if field == "slash_commands"
1079
- else KNOWN_SAFE_BUILTIN_SKILLS
1080
- )
1081
- value = init_event.get(field)
1082
- if not isinstance(value, list) or any(not isinstance(item, str) for item in value):
1083
- return None, f"Claude host-vocabulary baseline has malformed {field}"
1084
- identifiers = [customization_entry_identifier(item) for item in value]
1085
- if (
1086
- len(identifiers) != len(set(identifiers))
1087
- or any(
1088
- not host_entry_is_whole(item, identifier)
1089
- for item, identifier in zip(value, identifiers)
1090
- )
1091
- or any(
1092
- identifier not in known_host_identifiers
1093
- and not is_bare_host_identifier(identifier)
1094
- for identifier in identifiers
1095
- )
1096
- ):
1097
- return None, f"Claude host-vocabulary baseline has unsafe {field}"
1098
- surfaces[field] = tuple(identifiers)
1099
-
1100
- terminal_commands = init_event.get("terminal_slash_commands", [])
1101
- if not isinstance(terminal_commands, list) or any(
1102
- not isinstance(item, str) for item in terminal_commands
1103
- ):
1104
- return None, "Claude host-vocabulary baseline has malformed terminal commands"
1105
- terminal_identifiers = [
1106
- customization_entry_identifier(item) for item in terminal_commands
1107
- ]
1108
- if (
1109
- len(terminal_identifiers) != len(set(terminal_identifiers))
1110
- or any(
1111
- not host_entry_is_whole(item, identifier)
1112
- for item, identifier in zip(terminal_commands, terminal_identifiers)
1113
- )
1114
- or not set(terminal_identifiers).issubset(set(surfaces["slash_commands"]))
1115
- ):
1116
- return None, "Claude host-vocabulary baseline has unsafe terminal commands"
1117
-
1118
- known_fields = {
1119
- "tools",
1120
- "capabilities",
1121
- *REQUIRED_EMPTY_INIT_FIELDS,
1122
- *KNOWN_SAFE_INIT_METADATA_FIELDS,
1123
- *KNOWN_SAFE_INIT_LIST_METADATA_FIELDS,
1124
- }
1125
- for field, value in init_event.items():
1126
- if field not in known_fields and is_authority_name(field):
1127
- return None, "Claude host-vocabulary baseline has unverifiable authority"
1128
- if field not in known_fields and isinstance(value, (list, dict)) and value:
1129
- return None, "Claude host-vocabulary baseline has an unknown non-empty surface"
1130
- if field in ("agents", "capabilities") and (
1131
- not isinstance(value, list)
1132
- or any(not isinstance(item, str) for item in value)
1133
- ):
1134
- return None, f"Claude host-vocabulary baseline has malformed {field}"
1135
-
1136
- return {
1137
- "claude_code_version": version,
1138
- "slash_commands": surfaces["slash_commands"],
1139
- "skills": surfaces["skills"],
1140
- }, None
1141
-
1142
-
1143
- def unexpected_customization_identifiers(
1144
- events: list[dict],
1145
- expected_native_skills: set[str] | None = None,
1146
- host_init_baseline: dict[str, object] | None = None,
1147
- ) -> set[str]:
1148
- """Return bounded identifiers without exposing customization bodies."""
1149
- expected_native_skills = expected_native_skills or set()
1150
- baseline_commands = (
1151
- set(host_init_baseline.get("slash_commands", ()))
1152
- if host_init_baseline
1153
- else set()
1154
- )
1155
- baseline_skills = (
1156
- set(host_init_baseline.get("skills", ()))
1157
- if host_init_baseline
1158
- else set()
1159
698
  )
1160
- unexpected: set[str] = set()
1161
- for ev in events:
1162
- if ev.get("type") != "system" or ev.get("subtype") != "init":
1163
- continue
1164
- for field in ("slash_commands", "skills", "plugins"):
1165
- value = ev.get(field)
1166
- if not isinstance(value, list):
1167
- continue
1168
- for entry in value:
1169
- if customization_entry_allowed(
1170
- field,
1171
- entry,
1172
- expected_native_skills,
1173
- baseline_commands,
1174
- baseline_skills,
1175
- ):
1176
- continue
1177
- unexpected.add(
1178
- f"{field}:{customization_entry_identifier(entry)}"
1179
- )
1180
- return unexpected
1181
699
 
1182
700
 
1183
701
  def invoked_tool_names(events: list[dict]) -> set[str]:
@@ -1263,20 +781,11 @@ def stream_probe_passed(
1263
781
  expected_tools: set[str] | None = None,
1264
782
  allow_expected_tool_use: bool = False,
1265
783
  runtime_surface_only: bool = False,
1266
- expected_native_skills: set[str] | None = None,
1267
- required_native_skills: set[str] | None = None,
1268
- host_init_baseline: dict[str, object] | None = None,
1269
784
  ) -> bool:
1270
785
  """Ground-truth pass: all init capability surfaces are present and empty,
1271
786
  no tool was invoked, and the result envelope is clean. Reply text is ignored."""
1272
787
  expected_tools = expected_tools or set()
1273
- state = init_surface_state(
1274
- events,
1275
- expected_tools,
1276
- expected_native_skills,
1277
- required_native_skills,
1278
- host_init_baseline,
1279
- )
788
+ state = init_surface_state(events)
1280
789
  if state is None:
1281
790
  return False
1282
791
  (
@@ -1286,21 +795,17 @@ def stream_probe_passed(
1286
795
  unknown_fields,
1287
796
  unsafe_values,
1288
797
  unverifiable_authority,
1289
- unclassifiable_vocabulary,
1290
798
  ) = state
1291
799
  invoked = invoked_tool_names(events)
1292
- # Every state set refuses here, including the unclassifiable-vocabulary one.
1293
- # That is the whole safety argument for reclassifying it: the softer class
1294
- # changes which client serves the review, never whether an unverified
1295
- # isolation surface is ACCEPTED. There is no path from a bare unrecognised
1296
- # name to a passing probe.
800
+ # Every state set refuses here: the softer classes change which client
801
+ # serves the review, never whether an unverified isolation surface is
802
+ # ACCEPTED.
1297
803
  if (
1298
804
  missing_or_invalid
1299
805
  or nonempty
1300
806
  or unknown_fields
1301
807
  or unsafe_values
1302
808
  or unverifiable_authority
1303
- or unclassifiable_vocabulary
1304
809
  or declared_tools != expected_tools
1305
810
  or invoked - expected_tools
1306
811
  or (invoked and not allow_expected_tool_use)
@@ -1333,9 +838,6 @@ def classify_failure(
1333
838
  stderr: str,
1334
839
  expected_tools: set[str] | None = None,
1335
840
  allow_expected_tool_use: bool = False,
1336
- expected_native_skills: set[str] | None = None,
1337
- required_native_skills: set[str] | None = None,
1338
- host_init_baseline: dict[str, object] | None = None,
1339
841
  ):
1340
842
  expected_tools = expected_tools or set()
1341
843
  check_label = "Claude runtime isolation check" if expected_tools else "Claude no-tool probe"
@@ -1440,13 +942,7 @@ def classify_failure(
1440
942
  "tool-availability ground-truth is unverifiable",
1441
943
  False,
1442
944
  )
1443
- state = init_surface_state(
1444
- failure_events,
1445
- expected_tools,
1446
- expected_native_skills,
1447
- required_native_skills,
1448
- host_init_baseline,
1449
- )
945
+ state = init_surface_state(failure_events)
1450
946
  if state is None:
1451
947
  return (
1452
948
  f"{check_label} stream-json capture is missing the init event; "
@@ -1460,7 +956,6 @@ def classify_failure(
1460
956
  unknown_fields,
1461
957
  unsafe_values,
1462
958
  unverifiable_authority,
1463
- unclassifiable_vocabulary,
1464
959
  ) = state
1465
960
  if missing_or_invalid:
1466
961
  return (
@@ -1523,42 +1018,16 @@ def classify_failure(
1523
1018
  "verify isolation",
1524
1019
  False,
1525
1020
  )
1526
- if unclassifiable_vocabulary and not surface_breached:
1527
- # Same guard, same reason as the two branches above: reached only
1528
- # when nothing is actually breached, so a hostile CLI cannot use a
1529
- # bare name to launder a real breach into "try another client".
1530
- #
1531
- # A distinct phrase, not a reuse of the surface-shaped one: what is
1532
- # unrecognised here is an IDENTIFIER, not a field, and describing it
1533
- # as the wrong thing is a diagnostic this repo has already paid for.
1534
- # `claude_review.sh` carries a matching arm ahead of its terminal
1535
- # arm; relying on its late `*"init"*` catch-all would make routing
1536
- # depend on `case`-arm order alone.
1537
- return (
1538
- "Claude reviewer lane found an unclassifiable host-vocabulary "
1539
- "entry ("
1540
- + ", ".join(
1541
- safe_identifier(v) for v in sorted(unclassifiable_vocabulary)
1542
- )
1543
- + "); the built-in allowlist cannot prove whether the host or a "
1544
- "user owns that name, so this lane cannot verify isolation",
1545
- False,
1546
- )
1547
- if (
1548
- surface_breached
1549
- or unknown_fields
1550
- or unverifiable_authority
1551
- or unclassifiable_vocabulary
1552
- ):
1021
+ if surface_breached or unknown_fields or unverifiable_authority:
1553
1022
  if expected_tools:
1554
1023
  surface_reason = (
1555
1024
  f"{check_label} runtime capability surface does not match the expected boundary; "
1556
- "a tool, MCP server, skill, command, or plugin is unexpected or unsafe"
1025
+ "a tool or MCP server is unexpected or unsafe"
1557
1026
  )
1558
1027
  else:
1559
1028
  surface_reason = (
1560
1029
  "Claude no-tool probe runtime capability surface is not empty; "
1561
- "a tool, MCP server, skill, command, or plugin remains declared or invoked"
1030
+ "a tool or MCP server remains declared or invoked"
1562
1031
  )
1563
1032
  return (
1564
1033
  surface_reason,
@@ -1573,21 +1042,14 @@ def main() -> int:
1573
1042
  print(
1574
1043
  "usage: parse_probe_result.py RC STDOUT_FILE STDERR_FILE "
1575
1044
  "[--require-empty-init] [--expected-tools CSV] "
1576
- "[--expected-native-skills CSV] "
1577
- "[--required-native-skills CSV] "
1578
- "[--host-init-baseline FILE] [--validate-host-init-baseline] "
1579
1045
  "[--allow-expected-tool-use] [--runtime-surface-only]",
1580
1046
  file=sys.stderr,
1581
1047
  )
1582
1048
  return 64
1583
1049
  require_empty_init = False
1584
1050
  expected_tools: set[str] = set()
1585
- expected_native_skills: set[str] = set()
1586
- required_native_skills: set[str] = set()
1587
1051
  allow_expected_tool_use = False
1588
1052
  runtime_surface_only = False
1589
- host_init_baseline_path = ""
1590
- validate_host_init_baseline = False
1591
1053
  option_index = 4
1592
1054
  while option_index < len(sys.argv):
1593
1055
  option = sys.argv[option_index]
@@ -1601,32 +1063,12 @@ def main() -> int:
1601
1063
  if name.strip()
1602
1064
  }
1603
1065
  option_index += 2
1604
- elif option == "--expected-native-skills" and option_index + 1 < len(sys.argv):
1605
- expected_native_skills = {
1606
- name.strip().lower()
1607
- for name in sys.argv[option_index + 1].split(",")
1608
- if name.strip()
1609
- }
1610
- option_index += 2
1611
- elif option == "--required-native-skills" and option_index + 1 < len(sys.argv):
1612
- required_native_skills = {
1613
- name.strip().lower()
1614
- for name in sys.argv[option_index + 1].split(",")
1615
- if name.strip()
1616
- }
1617
- option_index += 2
1618
1066
  elif option == "--allow-expected-tool-use":
1619
1067
  allow_expected_tool_use = True
1620
1068
  option_index += 1
1621
1069
  elif option == "--runtime-surface-only":
1622
1070
  runtime_surface_only = True
1623
1071
  option_index += 1
1624
- elif option == "--host-init-baseline" and option_index + 1 < len(sys.argv):
1625
- host_init_baseline_path = sys.argv[option_index + 1]
1626
- option_index += 2
1627
- elif option == "--validate-host-init-baseline":
1628
- validate_host_init_baseline = True
1629
- option_index += 1
1630
1072
  else:
1631
1073
  print("unknown or incomplete probe parser option", file=sys.stderr)
1632
1074
  return 64
@@ -1634,13 +1076,7 @@ def main() -> int:
1634
1076
  # Tool-boundary verdicts are meaningful only with stream-json init
1635
1077
  # evidence. Make the stronger requirement intrinsic so a direct caller
1636
1078
  # cannot accidentally re-enable the legacy text fallback.
1637
- if (
1638
- runtime_surface_only
1639
- or expected_tools
1640
- or allow_expected_tool_use
1641
- or expected_native_skills
1642
- or required_native_skills
1643
- ):
1079
+ if runtime_surface_only or expected_tools or allow_expected_tool_use:
1644
1080
  require_empty_init = True
1645
1081
 
1646
1082
  try:
@@ -1651,20 +1087,6 @@ def main() -> int:
1651
1087
 
1652
1088
  stdout = Path(sys.argv[2]).read_text(encoding="utf-8", errors="replace")
1653
1089
  stderr = Path(sys.argv[3]).read_text(encoding="utf-8", errors="replace")
1654
- host_init_baseline = None
1655
- if validate_host_init_baseline:
1656
- host_init_baseline, baseline_error = load_host_init_baseline(sys.argv[2])
1657
- if baseline_error is not None:
1658
- print(json.dumps({"reason": baseline_error, "generic_failure": False}))
1659
- return 1
1660
- return 0
1661
- if host_init_baseline_path:
1662
- host_init_baseline, baseline_error = load_host_init_baseline(
1663
- host_init_baseline_path
1664
- )
1665
- if baseline_error is not None:
1666
- print(json.dumps({"reason": baseline_error, "generic_failure": False}))
1667
- return 1
1668
1090
  normalized_stdout = normalize(stdout)
1669
1091
  stderr_surface = classification_surface(None, "", stderr)
1670
1092
  env = parse_json_object(normalized_stdout)
@@ -1714,9 +1136,6 @@ def main() -> int:
1714
1136
  expected_tools,
1715
1137
  allow_expected_tool_use,
1716
1138
  runtime_surface_only,
1717
- expected_native_skills,
1718
- required_native_skills,
1719
- host_init_baseline,
1720
1139
  ):
1721
1140
  return 0
1722
1141
  elif not require_empty_init:
@@ -1767,13 +1186,7 @@ def main() -> int:
1767
1186
  f"{snippet or 'rate limit or quota exceeded'}"
1768
1187
  )
1769
1188
  else:
1770
- runtime_state = init_surface_state(
1771
- events,
1772
- expected_tools,
1773
- expected_native_skills,
1774
- required_native_skills,
1775
- host_init_baseline,
1776
- )
1189
+ runtime_state = init_surface_state(events)
1777
1190
  runtime_detail = ""
1778
1191
  runtime_drift_only = False
1779
1192
  if runtime_state is not None:
@@ -1784,23 +1197,12 @@ def main() -> int:
1784
1197
  unknown,
1785
1198
  unsafe,
1786
1199
  unverifiable,
1787
- vocabulary,
1788
1200
  ) = runtime_state
1789
1201
  # Schema drift routes the same way on the main-invocation path as
1790
1202
  # on the probe path. Without this, an unrecognized init container
1791
1203
  # would terminate the lane here while merely falling back there —
1792
1204
  # the same condition, two verdicts, decided by which parse ran.
1793
- #
1794
- # `unexpected_identifiers` is deliberately absent from the guard,
1795
- # but NOT because it is redundant with `nonempty` — that was true
1796
- # only while every unrecognised entry was a breach. An entry it
1797
- # reports now lands in either `nonempty` (proven customization)
1798
- # or `vocabulary` (unclassifiable host name), both of which ARE
1799
- # in the guard; it stays out because it would double-count them
1800
- # while distinguishing neither. Fixtures pin the combined
1801
- # drift+breach and vocabulary+breach cases so this stays true
1802
- # rather than being taken on faith.
1803
- runtime_drift_only = bool(unknown or unverifiable or vocabulary) and not (
1205
+ runtime_drift_only = bool(unknown or unverifiable) and not (
1804
1206
  missing
1805
1207
  or nonempty
1806
1208
  or unsafe
@@ -1808,15 +1210,12 @@ def main() -> int:
1808
1210
  or invoked_tools - expected_tools
1809
1211
  or (invoked_tools and not allow_expected_tool_use)
1810
1212
  )
1811
- unexpected_identifiers = unexpected_customization_identifiers(
1812
- events, expected_native_skills, host_init_baseline
1813
- )
1814
1213
  # EVERY interpolated identifier is CLI-supplied, not just the
1815
- # field names: tool names, skill/plugin/command identifiers and
1816
- # invoked-tool names come from the inspected stream too, and may
1817
- # contain spaces just as freely. Any one of them left raw lets
1818
- # the inspected CLI pick which routing arm matches — a breach
1819
- # laundered into "try another client".
1214
+ # field names: tool names and invoked-tool names come from the
1215
+ # inspected stream too, and may contain spaces just as freely.
1216
+ # Any one of them left raw lets the inspected CLI pick which
1217
+ # routing arm matches — a breach laundered into "try another
1218
+ # client".
1820
1219
  def joined(values):
1821
1220
  return ",".join(safe_identifier(v) for v in sorted(values))
1822
1221
 
@@ -1825,8 +1224,6 @@ def main() -> int:
1825
1224
  + joined(missing)
1826
1225
  + "; unexpected_customizations="
1827
1226
  + joined(nonempty)
1828
- + "; unexpected_customization_identifiers="
1829
- + joined(unexpected_identifiers)
1830
1227
  + "; declared_tools="
1831
1228
  + joined(declared)
1832
1229
  + "; invoked_tools="
@@ -1837,22 +1234,8 @@ def main() -> int:
1837
1234
  + joined(unsafe)
1838
1235
  + "; unverifiable_authority="
1839
1236
  + joined(unverifiable)
1840
- + "; unclassifiable_host_vocabulary="
1841
- + joined(vocabulary)
1842
- )
1843
- if runtime_drift_only and not (unknown or unverifiable):
1844
- # The new class gets its own phrase only when it is the SOLE
1845
- # unverifiable finding. Combined with schema drift or an
1846
- # unverifiable authority knob, the existing phrase still applies
1847
- # and still routes to the same fallback-eligible arm, so the
1848
- # wording of every pre-existing case is left exactly as it was.
1849
- runtime_reason = (
1850
- "Claude main invocation found an unclassifiable "
1851
- "host-vocabulary entry; the built-in allowlist cannot prove "
1852
- "whether the host or a user owns that name, so this reviewer "
1853
- "lane cannot verify isolation" + runtime_detail
1854
1237
  )
1855
- elif runtime_drift_only:
1238
+ if runtime_drift_only:
1856
1239
  runtime_reason = (
1857
1240
  "Claude main invocation found an unrecognized surface-shaped "
1858
1241
  "init field; the stream-json schema must be reviewed before "
@@ -1879,9 +1262,6 @@ def main() -> int:
1879
1262
  stderr,
1880
1263
  expected_tools,
1881
1264
  allow_expected_tool_use,
1882
- expected_native_skills,
1883
- required_native_skills,
1884
- host_init_baseline,
1885
1265
  )
1886
1266
  print(
1887
1267
  json.dumps(