@ccoalm/ccl-skills 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
- package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
- package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/SKILL.md +14 -41
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/attention-budget-ratchet.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/correction-routing-map.md +22 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/coverage-exhaustion-traps.md +7 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/description-authoring.md +26 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/dual-track-review-gate.md +2 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +7 -1
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/extraction-quickstart.md +4 -2
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/harness-patterns-and-eval.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/incident-postmortem-extraction.md +8 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/rule-consolidation.md +1 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +73 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/uiux-judgment-extraction.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/validation-and-landing.md +11 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/entrypoint_form_census.py +169 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/reference-access-census.sh +157 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +483 -109
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +10 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_entrypoint_form_census.sh +174 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_reference_access_census.sh +209 -0
- package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +394 -5
- package/dist/assets/release.json +77 -47
- package/package.json +1 -1
package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py
CHANGED
|
@@ -334,11 +334,13 @@ def parse_json_object(text: str) -> object | None:
|
|
|
334
334
|
# block may occur. Ground truth comes from the stream, not the model's reply
|
|
335
335
|
# token, which can hallucinate TOOL_ENABLED even when no tool exists.
|
|
336
336
|
PROBE_TOOL = "bash"
|
|
337
|
+
# The only customization surface still required to be EMPTY. An inherited MCP
|
|
338
|
+
# server is invocable capability, so it is a breach. Skill, command and plugin
|
|
339
|
+
# lists are host/plugin VOCABULARY: nothing in them is invocable while `tools`
|
|
340
|
+
# is pinned to the expected set and the tool_use scan is clean, so they are
|
|
341
|
+
# recorded metadata and never judged -- see KNOWN_VOCABULARY_INIT_FIELDS.
|
|
337
342
|
REQUIRED_EMPTY_INIT_FIELDS = (
|
|
338
343
|
"mcp_servers",
|
|
339
|
-
"slash_commands",
|
|
340
|
-
"skills",
|
|
341
|
-
"plugins",
|
|
342
344
|
)
|
|
343
345
|
# Fields that must be PRESENT (not merely empty). `permissionMode` is the only
|
|
344
346
|
# report of the reviewer's authority, and authority is scalar-shaped — which is
|
|
@@ -444,90 +446,21 @@ KNOWN_SAFE_INIT_METADATA_FIELDS = {
|
|
|
444
446
|
"uuid",
|
|
445
447
|
"fast_mode_state",
|
|
446
448
|
}
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
449
|
+
# Host/plugin vocabulary the init event enumerates. Deliberately NOT a
|
|
450
|
+
# boundary and never judged by name, shape, or origin: an entry here becomes
|
|
451
|
+
# invocable only through a tool, and the exact `tools` allowlist plus the
|
|
452
|
+
# tool_use scan already pin that. Judging these lists against a snapshot of the
|
|
453
|
+
# host's own built-in names turned every CLI release that shipped a new skill or
|
|
454
|
+
# command into a reviewer-lane outage while proving nothing about isolation, and
|
|
455
|
+
# it made an uninstalled or older plugin a refusal instead of a review without
|
|
456
|
+
# owner skills.
|
|
457
|
+
KNOWN_VOCABULARY_INIT_FIELDS = (
|
|
458
|
+
"slash_commands",
|
|
452
459
|
"terminal_slash_commands",
|
|
453
|
-
|
|
460
|
+
"skills",
|
|
461
|
+
"plugins",
|
|
462
|
+
)
|
|
454
463
|
KNOWN_SAFE_PERMISSION_MODES = {"default", "plan"}
|
|
455
|
-
KNOWN_SAFE_BUILTIN_SKILLS = {
|
|
456
|
-
"batch",
|
|
457
|
-
"claude-api",
|
|
458
|
-
"code-review",
|
|
459
|
-
"dataviz",
|
|
460
|
-
"debug",
|
|
461
|
-
"deep-research",
|
|
462
|
-
"design-sync",
|
|
463
|
-
"doctor",
|
|
464
|
-
"fewer-permission-prompts",
|
|
465
|
-
"loop",
|
|
466
|
-
"run",
|
|
467
|
-
"run-skill-generator",
|
|
468
|
-
"schedule",
|
|
469
|
-
"simplify",
|
|
470
|
-
"update-config",
|
|
471
|
-
"verify",
|
|
472
|
-
}
|
|
473
|
-
KNOWN_SAFE_BUILTIN_SLASH_COMMANDS = {
|
|
474
|
-
"__remote-workflow",
|
|
475
|
-
"agents",
|
|
476
|
-
"batch",
|
|
477
|
-
"claude-api",
|
|
478
|
-
"clear",
|
|
479
|
-
"code-review",
|
|
480
|
-
"color",
|
|
481
|
-
"compact",
|
|
482
|
-
"config",
|
|
483
|
-
"context",
|
|
484
|
-
"dataviz",
|
|
485
|
-
"debug",
|
|
486
|
-
"deep-research",
|
|
487
|
-
"design",
|
|
488
|
-
"design-consent",
|
|
489
|
-
"design-revoke",
|
|
490
|
-
"design-sync",
|
|
491
|
-
"doctor",
|
|
492
|
-
"effort",
|
|
493
|
-
"extra-usage",
|
|
494
|
-
"fast",
|
|
495
|
-
"fewer-permission-prompts",
|
|
496
|
-
"goal",
|
|
497
|
-
"heapdump",
|
|
498
|
-
# `/import` arrived in a CLI release after this snapshot was written, which
|
|
499
|
-
# is how the snapshot's cost was discovered: an unrecognised name used to be
|
|
500
|
-
# classified as a proven customization -- terminal and non-cascadable -- so
|
|
501
|
-
# one new built-in took out the whole reviewer lane. That is fixed at the
|
|
502
|
-
# CLASS level now (see is_bare_host_identifier): a bare name this list does
|
|
503
|
-
# not know is unverifiable and cascades. This list therefore no longer
|
|
504
|
-
# decides whether review is POSSIBLE, only whether the Claude lane is the
|
|
505
|
-
# one that serves it, so a missing name costs a client switch instead of the
|
|
506
|
-
# capability. Keep it accurate where it is cheap; do not treat adding a name
|
|
507
|
-
# as the fix for a lane outage.
|
|
508
|
-
"import",
|
|
509
|
-
"init",
|
|
510
|
-
"insights",
|
|
511
|
-
"loop",
|
|
512
|
-
"mcp",
|
|
513
|
-
"model",
|
|
514
|
-
"recap",
|
|
515
|
-
"reload-skills",
|
|
516
|
-
"rename",
|
|
517
|
-
"review",
|
|
518
|
-
"run",
|
|
519
|
-
"run-skill-generator",
|
|
520
|
-
"schedule",
|
|
521
|
-
"security-review",
|
|
522
|
-
"simplify",
|
|
523
|
-
"team-onboarding",
|
|
524
|
-
"ultrareview",
|
|
525
|
-
"update-config",
|
|
526
|
-
"usage",
|
|
527
|
-
"usage-credits",
|
|
528
|
-
"verify",
|
|
529
|
-
"workflow-launch-exec",
|
|
530
|
-
}
|
|
531
464
|
|
|
532
465
|
|
|
533
466
|
def stream_events(text: str) -> list[dict]:
|
|
@@ -644,22 +577,16 @@ def init_declared_tools(events: list[dict]) -> set[str] | None:
|
|
|
644
577
|
|
|
645
578
|
def init_surface_state(
|
|
646
579
|
events: list[dict],
|
|
647
|
-
|
|
648
|
-
expected_native_skills: set[str] | None = None,
|
|
649
|
-
required_native_skills: set[str] | None = None,
|
|
650
|
-
host_init_baseline: dict[str, object] | None = None,
|
|
651
|
-
) -> tuple[
|
|
652
|
-
set[str], set[str], set[str], set[str], set[str], set[str], set[str]
|
|
653
|
-
] | None:
|
|
580
|
+
) -> tuple[set[str], set[str], set[str], set[str], set[str], set[str]] | None:
|
|
654
581
|
"""Return init isolation state across all init events.
|
|
655
582
|
|
|
656
|
-
Result fields: missing/invalid required fields, non-empty
|
|
583
|
+
Result fields: missing/invalid required fields, non-empty capability
|
|
657
584
|
fields, declared tool names, unrecognized surface-shaped fields, fields
|
|
658
|
-
declaring an unsafe value, fields whose authority is unverifiable
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
585
|
+
declaring an unsafe value, and fields whose authority is unverifiable. A
|
|
586
|
+
missing or wrongly typed required field is not treated as empty, because
|
|
587
|
+
that would turn a CLI schema change into a silent gate bypass. Unknown
|
|
588
|
+
*scalar* metadata is tolerated on purpose — see the drift policy on
|
|
589
|
+
KNOWN_SAFE_INIT_METADATA_FIELDS.
|
|
663
590
|
|
|
664
591
|
Unsafe *values* are kept apart from unrecognized *shapes* on purpose: an
|
|
665
592
|
unknown schema addition means this lane cannot verify isolation (report it,
|
|
@@ -668,31 +595,11 @@ def init_surface_state(
|
|
|
668
595
|
must terminate the lane. Merging them would let a real privilege escalation
|
|
669
596
|
inherit the drift class's softer next_action.
|
|
670
597
|
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
unparseable entry IS a customization and stays in `nonempty`.
|
|
598
|
+
Skill, command and plugin lists are vocabulary, not capability: they are
|
|
599
|
+
read for nothing and may hold any value. Isolation is proven by the exact
|
|
600
|
+
`tools` allowlist, the tool_use scan, the empty MCP list and the pinned
|
|
601
|
+
`permissionMode` — none of which any vocabulary entry can affect.
|
|
676
602
|
"""
|
|
677
|
-
expected_native_skills = expected_native_skills or set()
|
|
678
|
-
required_native_skills = {
|
|
679
|
-
name.lower() for name in (required_native_skills or set())
|
|
680
|
-
}
|
|
681
|
-
baseline_commands = (
|
|
682
|
-
set(host_init_baseline.get("slash_commands", ()))
|
|
683
|
-
if host_init_baseline
|
|
684
|
-
else set()
|
|
685
|
-
)
|
|
686
|
-
baseline_skills = (
|
|
687
|
-
set(host_init_baseline.get("skills", ()))
|
|
688
|
-
if host_init_baseline
|
|
689
|
-
else set()
|
|
690
|
-
)
|
|
691
|
-
baseline_version = (
|
|
692
|
-
host_init_baseline.get("claude_code_version")
|
|
693
|
-
if host_init_baseline
|
|
694
|
-
else None
|
|
695
|
-
)
|
|
696
603
|
saw_init = False
|
|
697
604
|
missing_or_invalid: set[str] = set()
|
|
698
605
|
nonempty: set[str] = set()
|
|
@@ -706,23 +613,12 @@ def init_surface_state(
|
|
|
706
613
|
# change exists to remove. A KNOWN field carrying a KNOWN-unsafe value stays
|
|
707
614
|
# terminal in `unsafe_values`, because that one is proven.
|
|
708
615
|
unverifiable_authority: set[str] = set()
|
|
709
|
-
# Host vocabulary this snapshot does not recognise. Kept apart from
|
|
710
|
-
# `nonempty` for the same reason `unverifiable_authority` is: reporting an
|
|
711
|
-
# unverifiable thing in the proven-breach class is what turned a routine CLI
|
|
712
|
-
# release into a total review outage. Bare identifiers only — see
|
|
713
|
-
# is_bare_host_identifier for what stays terminal.
|
|
714
|
-
unclassifiable_vocabulary: set[str] = set()
|
|
715
|
-
declared_customizations: dict[str, set[str]] = {
|
|
716
|
-
"slash_commands": set(),
|
|
717
|
-
"skills": set(),
|
|
718
|
-
"plugins": set(),
|
|
719
|
-
}
|
|
720
616
|
known_fields = {
|
|
721
617
|
"tools",
|
|
722
618
|
"capabilities",
|
|
723
619
|
*REQUIRED_EMPTY_INIT_FIELDS,
|
|
620
|
+
*KNOWN_VOCABULARY_INIT_FIELDS,
|
|
724
621
|
*KNOWN_SAFE_INIT_METADATA_FIELDS,
|
|
725
|
-
*KNOWN_SAFE_INIT_LIST_METADATA_FIELDS,
|
|
726
622
|
}
|
|
727
623
|
for ev in events:
|
|
728
624
|
if ev.get("type") != "system" or ev.get("subtype") != "init":
|
|
@@ -744,49 +640,6 @@ def init_surface_state(
|
|
|
744
640
|
value = ev.get(field)
|
|
745
641
|
if not isinstance(value, list):
|
|
746
642
|
missing_or_invalid.add(field)
|
|
747
|
-
elif field in declared_customizations:
|
|
748
|
-
identifiers = [
|
|
749
|
-
customization_entry_identifier(entry) for entry in value
|
|
750
|
-
]
|
|
751
|
-
declared_customizations[field].update(identifiers)
|
|
752
|
-
if len(identifiers) != len(set(identifiers)):
|
|
753
|
-
nonempty.add(field)
|
|
754
|
-
if value and expected_native_skills:
|
|
755
|
-
for entry in value:
|
|
756
|
-
identifier = customization_entry_identifier(entry)
|
|
757
|
-
if field in HOST_VOCABULARY_FIELDS and (
|
|
758
|
-
not host_entry_is_whole(entry, identifier)
|
|
759
|
-
):
|
|
760
|
-
# Nothing may be discarded to reach a verdict on
|
|
761
|
-
# these fields — see host_entry_is_whole. Runs BEFORE
|
|
762
|
-
# the allowlist, because the allowlist consumes the
|
|
763
|
-
# same lossy identifier and would clear the entry
|
|
764
|
-
# outright on a truncated or dict-supplied name.
|
|
765
|
-
# Legitimate namespaced entries are unaffected: their
|
|
766
|
-
# whole value IS the identifier.
|
|
767
|
-
nonempty.add(field)
|
|
768
|
-
continue
|
|
769
|
-
if customization_entry_allowed(
|
|
770
|
-
field,
|
|
771
|
-
entry,
|
|
772
|
-
expected_native_skills,
|
|
773
|
-
baseline_commands,
|
|
774
|
-
baseline_skills,
|
|
775
|
-
):
|
|
776
|
-
continue
|
|
777
|
-
if field in HOST_VOCABULARY_FIELDS and (
|
|
778
|
-
is_bare_host_identifier(identifier)
|
|
779
|
-
):
|
|
780
|
-
# Cannot be shown to be a customization: the
|
|
781
|
-
# allowlist it failed is a snapshot of a vocabulary
|
|
782
|
-
# the host owns. Refuse, but stay cascadable.
|
|
783
|
-
unclassifiable_vocabulary.add(
|
|
784
|
-
f"{field}:{identifier}"
|
|
785
|
-
)
|
|
786
|
-
else:
|
|
787
|
-
nonempty.add(field)
|
|
788
|
-
elif value:
|
|
789
|
-
nonempty.add(field)
|
|
790
643
|
elif value:
|
|
791
644
|
nonempty.add(field)
|
|
792
645
|
# Per EVENT, not on the union across events: a later init that drops or
|
|
@@ -824,73 +677,17 @@ def init_surface_state(
|
|
|
824
677
|
elif field == "permissionMode":
|
|
825
678
|
if value not in KNOWN_SAFE_PERMISSION_MODES:
|
|
826
679
|
unsafe_values.add(field)
|
|
827
|
-
elif field in
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
832
|
-
else:
|
|
833
|
-
identifiers = [
|
|
834
|
-
customization_entry_identifier(item) for item in value
|
|
835
|
-
]
|
|
836
|
-
if (
|
|
837
|
-
len(identifiers) != len(set(identifiers))
|
|
838
|
-
or any(
|
|
839
|
-
not host_entry_is_whole(item, identifier)
|
|
840
|
-
for item, identifier in zip(value, identifiers)
|
|
841
|
-
)
|
|
842
|
-
or not set(identifiers).issubset(
|
|
843
|
-
declared_customizations["slash_commands"]
|
|
844
|
-
)
|
|
845
|
-
):
|
|
846
|
-
unknown_fields.add(field)
|
|
680
|
+
elif field in KNOWN_VOCABULARY_INIT_FIELDS:
|
|
681
|
+
# Vocabulary, not capability: any value, any shape. Whatever a
|
|
682
|
+
# CLI release or an installed plugin lists here cannot be
|
|
683
|
+
# invoked past the pinned `tools` set.
|
|
684
|
+
continue
|
|
847
685
|
elif field in KNOWN_SAFE_INIT_METADATA_FIELDS and isinstance(
|
|
848
686
|
value, (list, dict)
|
|
849
687
|
):
|
|
850
688
|
unknown_fields.add(field)
|
|
851
|
-
if baseline_version is not None and ev.get("claude_code_version") != baseline_version:
|
|
852
|
-
# The two invocations no longer prove one same-version host
|
|
853
|
-
# vocabulary snapshot. Refuse this lane, but treat the mismatch as
|
|
854
|
-
# capability drift rather than a proven tool/authority breach so a
|
|
855
|
-
# different reviewer may continue.
|
|
856
|
-
unknown_fields.add("claude_code_version:host-baseline-mismatch")
|
|
857
689
|
if not saw_init:
|
|
858
690
|
return None
|
|
859
|
-
if expected_native_skills:
|
|
860
|
-
if "ccl-skills" not in declared_customizations["plugins"]:
|
|
861
|
-
missing_or_invalid.add("plugins:ccl-skills")
|
|
862
|
-
ambiguous_required_skills = (
|
|
863
|
-
required_native_skills & KNOWN_SAFE_BUILTIN_SKILLS
|
|
864
|
-
)
|
|
865
|
-
for skill_name in ambiguous_required_skills:
|
|
866
|
-
missing_or_invalid.add(f"skills:ambiguous-selected-owner:{skill_name}")
|
|
867
|
-
expected_ccl_skill_identifiers = {
|
|
868
|
-
identifier
|
|
869
|
-
for skill_name in expected_native_skills
|
|
870
|
-
for identifier in (skill_name, f"ccl-skills:{skill_name}")
|
|
871
|
-
}
|
|
872
|
-
declared_ccl_skills = (
|
|
873
|
-
declared_customizations["skills"]
|
|
874
|
-
& expected_ccl_skill_identifiers
|
|
875
|
-
) - KNOWN_SAFE_BUILTIN_SKILLS
|
|
876
|
-
if declared_ccl_skills:
|
|
877
|
-
for skill_name in required_native_skills:
|
|
878
|
-
namespaced = f"ccl-skills:{skill_name}"
|
|
879
|
-
if not {
|
|
880
|
-
skill_name,
|
|
881
|
-
namespaced,
|
|
882
|
-
} & declared_customizations["skills"]:
|
|
883
|
-
missing_or_invalid.add(f"skills:{skill_name}")
|
|
884
|
-
declared_ccl_commands = {
|
|
885
|
-
identifier
|
|
886
|
-
for identifier in declared_customizations["slash_commands"]
|
|
887
|
-
if identifier.startswith("ccl-skills:")
|
|
888
|
-
}
|
|
889
|
-
if declared_ccl_commands:
|
|
890
|
-
for skill_name in required_native_skills:
|
|
891
|
-
namespaced = f"ccl-skills:{skill_name}"
|
|
892
|
-
if namespaced not in declared_customizations["slash_commands"]:
|
|
893
|
-
missing_or_invalid.add(f"slash_commands:{namespaced}")
|
|
894
691
|
return (
|
|
895
692
|
missing_or_invalid,
|
|
896
693
|
nonempty,
|
|
@@ -898,286 +695,7 @@ def init_surface_state(
|
|
|
898
695
|
unknown_fields,
|
|
899
696
|
unsafe_values,
|
|
900
697
|
unverifiable_authority,
|
|
901
|
-
unclassifiable_vocabulary,
|
|
902
|
-
)
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
def customization_entry_identifier(entry: object) -> str:
|
|
906
|
-
raw_identifier = ""
|
|
907
|
-
if isinstance(entry, str):
|
|
908
|
-
raw_identifier = entry.strip().split(maxsplit=1)[0]
|
|
909
|
-
elif isinstance(entry, dict):
|
|
910
|
-
for key in ("name", "command", "id"):
|
|
911
|
-
candidate = entry.get(key)
|
|
912
|
-
if isinstance(candidate, str) and candidate.strip():
|
|
913
|
-
raw_identifier = candidate.strip().split(maxsplit=1)[0]
|
|
914
|
-
break
|
|
915
|
-
if not re.fullmatch(r"[A-Za-z0-9_./:@+-]{1,120}", raw_identifier):
|
|
916
|
-
return "<unidentified>"
|
|
917
|
-
return raw_identifier.lower().lstrip("/")
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
def customization_entry_allowed(
|
|
921
|
-
field: str,
|
|
922
|
-
entry: object,
|
|
923
|
-
expected_native_skills: set[str],
|
|
924
|
-
baseline_commands: set[str] | None = None,
|
|
925
|
-
baseline_skills: set[str] | None = None,
|
|
926
|
-
) -> bool:
|
|
927
|
-
identifier = customization_entry_identifier(entry)
|
|
928
|
-
baseline_commands = baseline_commands or set()
|
|
929
|
-
# Baseline skills remain available to drift diagnostics, but never grant
|
|
930
|
-
# formal-run authority: a future safe-mode regression could otherwise turn
|
|
931
|
-
# one leaked user skill into an accepted callable capability.
|
|
932
|
-
baseline_skills = baseline_skills or set()
|
|
933
|
-
selected_names = {name.lower() for name in expected_native_skills}
|
|
934
|
-
selected_namespaced = {
|
|
935
|
-
f"ccl-skills:{name}" for name in selected_names
|
|
936
|
-
}
|
|
937
|
-
if field == "plugins":
|
|
938
|
-
return identifier == "ccl-skills"
|
|
939
|
-
if field == "skills":
|
|
940
|
-
return (
|
|
941
|
-
identifier in KNOWN_SAFE_BUILTIN_SKILLS
|
|
942
|
-
or identifier in selected_names
|
|
943
|
-
or identifier in selected_namespaced
|
|
944
|
-
)
|
|
945
|
-
if field == "slash_commands":
|
|
946
|
-
return (
|
|
947
|
-
identifier in KNOWN_SAFE_BUILTIN_SLASH_COMMANDS
|
|
948
|
-
or identifier in baseline_commands
|
|
949
|
-
or identifier in selected_namespaced
|
|
950
|
-
)
|
|
951
|
-
return False
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
# Fields whose entries, in a real run, come from the HOST's built-in vocabulary
|
|
955
|
-
# rather than from anything this repo installs: measured at CLI 2.1.220, the
|
|
956
|
-
# review-skill invocation reports 46 built-in commands and 16 built-in skills and
|
|
957
|
-
# nothing else. `plugins` is deliberately absent -- a plugin is user-installed by
|
|
958
|
-
# definition, and the expected set there is exactly the one this repo owns.
|
|
959
|
-
HOST_VOCABULARY_FIELDS = ("slash_commands", "skills")
|
|
960
|
-
# A bare identifier: no namespace, no path. Deliberately NOT "anything the
|
|
961
|
-
# allowlist missed" -- the discriminator has to be a property of the identifier
|
|
962
|
-
# itself, because the allowlist is the thing that cannot be trusted to be
|
|
963
|
-
# current.
|
|
964
|
-
BARE_HOST_IDENTIFIER = re.compile(r"[a-z0-9][a-z0-9_.-]*\Z")
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
def host_entry_is_whole(entry: object, identifier: str) -> bool:
|
|
968
|
-
"""True when nothing about a host-vocabulary entry was discarded to read it.
|
|
969
|
-
|
|
970
|
-
`customization_entry_identifier` is deliberately lossy — it keeps the first
|
|
971
|
-
whitespace-delimited token and reads a dict's `name` — which is right for a
|
|
972
|
-
DIAGNOSTIC string and wrong for a trust decision. Three review findings in
|
|
973
|
-
this slice were one shape: evidence present in the entry but never read. A
|
|
974
|
-
dict hid a path in a sibling key; a dict hid it under an allowed built-in
|
|
975
|
-
name; and `"brand-new evil-plugin:pwn"` truncated to `brand-new` while the
|
|
976
|
-
discarded suffix was the very proof of a customization.
|
|
977
|
-
|
|
978
|
-
So the classification consumes the WHOLE value: the entry must be a plain
|
|
979
|
-
string whose normalization — case folding and at most one leading `/`, the
|
|
980
|
-
only normalization the identifier helper itself applies — reproduces the
|
|
981
|
-
identifier exactly. Any remainder, any other shape, and the entry is a
|
|
982
|
-
customization this parser cannot clear. Checked BEFORE the allowlist,
|
|
983
|
-
because the allowlist reads the same truncated token.
|
|
984
|
-
"""
|
|
985
|
-
if not isinstance(entry, str):
|
|
986
|
-
return False
|
|
987
|
-
# Deliberately NO `.strip()`. The first version of this function called it,
|
|
988
|
-
# which reproduced inside the fix the exact lossiness the fix exists to
|
|
989
|
-
# reject: `"import "` stripped to `import`, compared equal to the identifier,
|
|
990
|
-
# was declared whole, and the allowlist then ACCEPTED it with isolation
|
|
991
|
-
# reported verified. Surrounding whitespace is part of the value, so an entry
|
|
992
|
-
# carrying any is not a plain host name.
|
|
993
|
-
normalized = entry.lower()
|
|
994
|
-
if normalized.startswith("/"):
|
|
995
|
-
normalized = normalized[1:]
|
|
996
|
-
return normalized == identifier
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
def is_bare_host_identifier(identifier: str) -> bool:
|
|
1000
|
-
"""True when a disallowed entry cannot be PROVEN to be a customization.
|
|
1001
|
-
|
|
1002
|
-
Only ever consulted for a PLAIN STRING entry — see the caller. A dict entry
|
|
1003
|
-
is a structured descriptor whose other keys this parser does not validate,
|
|
1004
|
-
so `{"name": "brand-new", "command": "/x/y"}` would otherwise be classified
|
|
1005
|
-
on its bare `name` while a sibling key carried path-shaped proof of a real
|
|
1006
|
-
customization. That is not the unclassifiable case this class is for: the
|
|
1007
|
-
evidence was present and merely unread. Measured against the real CLI, both
|
|
1008
|
-
host-vocabulary fields arrive as plain strings (only `plugins`, already
|
|
1009
|
-
excluded, uses dicts), so refusing dicts here costs nothing operationally.
|
|
1010
|
-
|
|
1011
|
-
`customization_entry_identifier` has already lowercased, stripped a leading
|
|
1012
|
-
`/`, and replaced anything it could not parse with `<unidentified>`. What is
|
|
1013
|
-
left is bare only if it carries no namespace separator and no path
|
|
1014
|
-
separator, so each of these stays a proven breach:
|
|
1015
|
-
|
|
1016
|
-
* `evil-plugin:pwn` -- a namespace proves a surface beyond the one
|
|
1017
|
-
expected plugin, which is a customization, not host vocabulary.
|
|
1018
|
-
* `dir/cmd` -- no host built-in is spelled as a path.
|
|
1019
|
-
* `<unidentified>` -- an entry that failed the charset or was not a
|
|
1020
|
-
string/dict is not evidence of anything, least of all of host origin.
|
|
1021
|
-
|
|
1022
|
-
Note what this does NOT claim: a bare name may still be a user-authored
|
|
1023
|
-
command. The point is that we cannot tell, and the unclassifiable case
|
|
1024
|
-
belongs in the class that refuses and cascades rather than the class that
|
|
1025
|
-
refuses and destroys the capability. Isolation itself is still proven only
|
|
1026
|
-
by the exact `tools` allowlist, the tool_use scan, and the value-pinned
|
|
1027
|
-
`permissionMode` -- none of which any identifier here can affect.
|
|
1028
|
-
"""
|
|
1029
|
-
return bool(BARE_HOST_IDENTIFIER.fullmatch(identifier))
|
|
1030
|
-
|
|
1031
|
-
|
|
1032
|
-
def load_host_init_baseline(path: str) -> tuple[dict[str, object] | None, str | None]:
|
|
1033
|
-
"""Read one safe-mode, no-plugin init capture as version-owned vocabulary.
|
|
1034
|
-
|
|
1035
|
-
The baseline is not a verdict and its model result is ignored. It may name
|
|
1036
|
-
host commands and skills only because the wrapper invocation disables
|
|
1037
|
-
setting sources, CLAUDE.md, auto-memory, MCP, tools, and plugins. Any tool,
|
|
1038
|
-
plugin, malformed identifier, authority drift, or unknown non-empty surface
|
|
1039
|
-
keeps the baseline unusable rather than laundering it into the main run.
|
|
1040
|
-
"""
|
|
1041
|
-
try:
|
|
1042
|
-
text = Path(path).read_text(encoding="utf-8", errors="replace")
|
|
1043
|
-
except OSError:
|
|
1044
|
-
return None, "Claude host-vocabulary baseline could not be read"
|
|
1045
|
-
events = stream_events(normalize(text))
|
|
1046
|
-
stream_like, stream_corrupt = stream_classification(text, events)
|
|
1047
|
-
init_events = [
|
|
1048
|
-
event
|
|
1049
|
-
for event in events
|
|
1050
|
-
if event.get("type") == "system" and event.get("subtype") == "init"
|
|
1051
|
-
]
|
|
1052
|
-
if not stream_like or stream_corrupt or len(init_events) != 1:
|
|
1053
|
-
return None, "Claude host-vocabulary baseline lacks one intact init event"
|
|
1054
|
-
if invoked_tool_names(events):
|
|
1055
|
-
return None, "Claude host-vocabulary baseline invoked a tool"
|
|
1056
|
-
init_event = init_events[0]
|
|
1057
|
-
if init_event.get("tools") != []:
|
|
1058
|
-
return None, "Claude host-vocabulary baseline exposed a tool"
|
|
1059
|
-
# Keep this set-derived rather than spelling today's MCP/plugin fields
|
|
1060
|
-
# twice. If the formal init adds another required-empty customization
|
|
1061
|
-
# surface, the baseline must prove it empty before its command vocabulary
|
|
1062
|
-
# can carry authority. Known metadata such as `agents` is deliberately not
|
|
1063
|
-
# in REQUIRED_EMPTY_INIT_FIELDS: with tools pinned empty it is descriptive,
|
|
1064
|
-
# not model-invocable.
|
|
1065
|
-
for field in REQUIRED_EMPTY_INIT_FIELDS:
|
|
1066
|
-
if field not in HOST_VOCABULARY_FIELDS and init_event.get(field) != []:
|
|
1067
|
-
return None, f"Claude host-vocabulary baseline exposed {field}"
|
|
1068
|
-
if init_event.get("permissionMode") != "default":
|
|
1069
|
-
return None, "Claude host-vocabulary baseline changed permission authority"
|
|
1070
|
-
version = init_event.get("claude_code_version")
|
|
1071
|
-
if not isinstance(version, str) or not version or len(version) > 120:
|
|
1072
|
-
return None, "Claude host-vocabulary baseline lacks a bounded CLI version"
|
|
1073
|
-
|
|
1074
|
-
surfaces: dict[str, tuple[str, ...]] = {}
|
|
1075
|
-
for field in HOST_VOCABULARY_FIELDS:
|
|
1076
|
-
known_host_identifiers = (
|
|
1077
|
-
KNOWN_SAFE_BUILTIN_SLASH_COMMANDS
|
|
1078
|
-
if field == "slash_commands"
|
|
1079
|
-
else KNOWN_SAFE_BUILTIN_SKILLS
|
|
1080
|
-
)
|
|
1081
|
-
value = init_event.get(field)
|
|
1082
|
-
if not isinstance(value, list) or any(not isinstance(item, str) for item in value):
|
|
1083
|
-
return None, f"Claude host-vocabulary baseline has malformed {field}"
|
|
1084
|
-
identifiers = [customization_entry_identifier(item) for item in value]
|
|
1085
|
-
if (
|
|
1086
|
-
len(identifiers) != len(set(identifiers))
|
|
1087
|
-
or any(
|
|
1088
|
-
not host_entry_is_whole(item, identifier)
|
|
1089
|
-
for item, identifier in zip(value, identifiers)
|
|
1090
|
-
)
|
|
1091
|
-
or any(
|
|
1092
|
-
identifier not in known_host_identifiers
|
|
1093
|
-
and not is_bare_host_identifier(identifier)
|
|
1094
|
-
for identifier in identifiers
|
|
1095
|
-
)
|
|
1096
|
-
):
|
|
1097
|
-
return None, f"Claude host-vocabulary baseline has unsafe {field}"
|
|
1098
|
-
surfaces[field] = tuple(identifiers)
|
|
1099
|
-
|
|
1100
|
-
terminal_commands = init_event.get("terminal_slash_commands", [])
|
|
1101
|
-
if not isinstance(terminal_commands, list) or any(
|
|
1102
|
-
not isinstance(item, str) for item in terminal_commands
|
|
1103
|
-
):
|
|
1104
|
-
return None, "Claude host-vocabulary baseline has malformed terminal commands"
|
|
1105
|
-
terminal_identifiers = [
|
|
1106
|
-
customization_entry_identifier(item) for item in terminal_commands
|
|
1107
|
-
]
|
|
1108
|
-
if (
|
|
1109
|
-
len(terminal_identifiers) != len(set(terminal_identifiers))
|
|
1110
|
-
or any(
|
|
1111
|
-
not host_entry_is_whole(item, identifier)
|
|
1112
|
-
for item, identifier in zip(terminal_commands, terminal_identifiers)
|
|
1113
|
-
)
|
|
1114
|
-
or not set(terminal_identifiers).issubset(set(surfaces["slash_commands"]))
|
|
1115
|
-
):
|
|
1116
|
-
return None, "Claude host-vocabulary baseline has unsafe terminal commands"
|
|
1117
|
-
|
|
1118
|
-
known_fields = {
|
|
1119
|
-
"tools",
|
|
1120
|
-
"capabilities",
|
|
1121
|
-
*REQUIRED_EMPTY_INIT_FIELDS,
|
|
1122
|
-
*KNOWN_SAFE_INIT_METADATA_FIELDS,
|
|
1123
|
-
*KNOWN_SAFE_INIT_LIST_METADATA_FIELDS,
|
|
1124
|
-
}
|
|
1125
|
-
for field, value in init_event.items():
|
|
1126
|
-
if field not in known_fields and is_authority_name(field):
|
|
1127
|
-
return None, "Claude host-vocabulary baseline has unverifiable authority"
|
|
1128
|
-
if field not in known_fields and isinstance(value, (list, dict)) and value:
|
|
1129
|
-
return None, "Claude host-vocabulary baseline has an unknown non-empty surface"
|
|
1130
|
-
if field in ("agents", "capabilities") and (
|
|
1131
|
-
not isinstance(value, list)
|
|
1132
|
-
or any(not isinstance(item, str) for item in value)
|
|
1133
|
-
):
|
|
1134
|
-
return None, f"Claude host-vocabulary baseline has malformed {field}"
|
|
1135
|
-
|
|
1136
|
-
return {
|
|
1137
|
-
"claude_code_version": version,
|
|
1138
|
-
"slash_commands": surfaces["slash_commands"],
|
|
1139
|
-
"skills": surfaces["skills"],
|
|
1140
|
-
}, None
|
|
1141
|
-
|
|
1142
|
-
|
|
1143
|
-
def unexpected_customization_identifiers(
|
|
1144
|
-
events: list[dict],
|
|
1145
|
-
expected_native_skills: set[str] | None = None,
|
|
1146
|
-
host_init_baseline: dict[str, object] | None = None,
|
|
1147
|
-
) -> set[str]:
|
|
1148
|
-
"""Return bounded identifiers without exposing customization bodies."""
|
|
1149
|
-
expected_native_skills = expected_native_skills or set()
|
|
1150
|
-
baseline_commands = (
|
|
1151
|
-
set(host_init_baseline.get("slash_commands", ()))
|
|
1152
|
-
if host_init_baseline
|
|
1153
|
-
else set()
|
|
1154
|
-
)
|
|
1155
|
-
baseline_skills = (
|
|
1156
|
-
set(host_init_baseline.get("skills", ()))
|
|
1157
|
-
if host_init_baseline
|
|
1158
|
-
else set()
|
|
1159
698
|
)
|
|
1160
|
-
unexpected: set[str] = set()
|
|
1161
|
-
for ev in events:
|
|
1162
|
-
if ev.get("type") != "system" or ev.get("subtype") != "init":
|
|
1163
|
-
continue
|
|
1164
|
-
for field in ("slash_commands", "skills", "plugins"):
|
|
1165
|
-
value = ev.get(field)
|
|
1166
|
-
if not isinstance(value, list):
|
|
1167
|
-
continue
|
|
1168
|
-
for entry in value:
|
|
1169
|
-
if customization_entry_allowed(
|
|
1170
|
-
field,
|
|
1171
|
-
entry,
|
|
1172
|
-
expected_native_skills,
|
|
1173
|
-
baseline_commands,
|
|
1174
|
-
baseline_skills,
|
|
1175
|
-
):
|
|
1176
|
-
continue
|
|
1177
|
-
unexpected.add(
|
|
1178
|
-
f"{field}:{customization_entry_identifier(entry)}"
|
|
1179
|
-
)
|
|
1180
|
-
return unexpected
|
|
1181
699
|
|
|
1182
700
|
|
|
1183
701
|
def invoked_tool_names(events: list[dict]) -> set[str]:
|
|
@@ -1263,20 +781,11 @@ def stream_probe_passed(
|
|
|
1263
781
|
expected_tools: set[str] | None = None,
|
|
1264
782
|
allow_expected_tool_use: bool = False,
|
|
1265
783
|
runtime_surface_only: bool = False,
|
|
1266
|
-
expected_native_skills: set[str] | None = None,
|
|
1267
|
-
required_native_skills: set[str] | None = None,
|
|
1268
|
-
host_init_baseline: dict[str, object] | None = None,
|
|
1269
784
|
) -> bool:
|
|
1270
785
|
"""Ground-truth pass: all init capability surfaces are present and empty,
|
|
1271
786
|
no tool was invoked, and the result envelope is clean. Reply text is ignored."""
|
|
1272
787
|
expected_tools = expected_tools or set()
|
|
1273
|
-
state = init_surface_state(
|
|
1274
|
-
events,
|
|
1275
|
-
expected_tools,
|
|
1276
|
-
expected_native_skills,
|
|
1277
|
-
required_native_skills,
|
|
1278
|
-
host_init_baseline,
|
|
1279
|
-
)
|
|
788
|
+
state = init_surface_state(events)
|
|
1280
789
|
if state is None:
|
|
1281
790
|
return False
|
|
1282
791
|
(
|
|
@@ -1286,21 +795,17 @@ def stream_probe_passed(
|
|
|
1286
795
|
unknown_fields,
|
|
1287
796
|
unsafe_values,
|
|
1288
797
|
unverifiable_authority,
|
|
1289
|
-
unclassifiable_vocabulary,
|
|
1290
798
|
) = state
|
|
1291
799
|
invoked = invoked_tool_names(events)
|
|
1292
|
-
# Every state set refuses here
|
|
1293
|
-
#
|
|
1294
|
-
#
|
|
1295
|
-
# isolation surface is ACCEPTED. There is no path from a bare unrecognised
|
|
1296
|
-
# name to a passing probe.
|
|
800
|
+
# Every state set refuses here: the softer classes change which client
|
|
801
|
+
# serves the review, never whether an unverified isolation surface is
|
|
802
|
+
# ACCEPTED.
|
|
1297
803
|
if (
|
|
1298
804
|
missing_or_invalid
|
|
1299
805
|
or nonempty
|
|
1300
806
|
or unknown_fields
|
|
1301
807
|
or unsafe_values
|
|
1302
808
|
or unverifiable_authority
|
|
1303
|
-
or unclassifiable_vocabulary
|
|
1304
809
|
or declared_tools != expected_tools
|
|
1305
810
|
or invoked - expected_tools
|
|
1306
811
|
or (invoked and not allow_expected_tool_use)
|
|
@@ -1333,9 +838,6 @@ def classify_failure(
|
|
|
1333
838
|
stderr: str,
|
|
1334
839
|
expected_tools: set[str] | None = None,
|
|
1335
840
|
allow_expected_tool_use: bool = False,
|
|
1336
|
-
expected_native_skills: set[str] | None = None,
|
|
1337
|
-
required_native_skills: set[str] | None = None,
|
|
1338
|
-
host_init_baseline: dict[str, object] | None = None,
|
|
1339
841
|
):
|
|
1340
842
|
expected_tools = expected_tools or set()
|
|
1341
843
|
check_label = "Claude runtime isolation check" if expected_tools else "Claude no-tool probe"
|
|
@@ -1440,13 +942,7 @@ def classify_failure(
|
|
|
1440
942
|
"tool-availability ground-truth is unverifiable",
|
|
1441
943
|
False,
|
|
1442
944
|
)
|
|
1443
|
-
state = init_surface_state(
|
|
1444
|
-
failure_events,
|
|
1445
|
-
expected_tools,
|
|
1446
|
-
expected_native_skills,
|
|
1447
|
-
required_native_skills,
|
|
1448
|
-
host_init_baseline,
|
|
1449
|
-
)
|
|
945
|
+
state = init_surface_state(failure_events)
|
|
1450
946
|
if state is None:
|
|
1451
947
|
return (
|
|
1452
948
|
f"{check_label} stream-json capture is missing the init event; "
|
|
@@ -1460,7 +956,6 @@ def classify_failure(
|
|
|
1460
956
|
unknown_fields,
|
|
1461
957
|
unsafe_values,
|
|
1462
958
|
unverifiable_authority,
|
|
1463
|
-
unclassifiable_vocabulary,
|
|
1464
959
|
) = state
|
|
1465
960
|
if missing_or_invalid:
|
|
1466
961
|
return (
|
|
@@ -1523,42 +1018,16 @@ def classify_failure(
|
|
|
1523
1018
|
"verify isolation",
|
|
1524
1019
|
False,
|
|
1525
1020
|
)
|
|
1526
|
-
if
|
|
1527
|
-
# Same guard, same reason as the two branches above: reached only
|
|
1528
|
-
# when nothing is actually breached, so a hostile CLI cannot use a
|
|
1529
|
-
# bare name to launder a real breach into "try another client".
|
|
1530
|
-
#
|
|
1531
|
-
# A distinct phrase, not a reuse of the surface-shaped one: what is
|
|
1532
|
-
# unrecognised here is an IDENTIFIER, not a field, and describing it
|
|
1533
|
-
# as the wrong thing is a diagnostic this repo has already paid for.
|
|
1534
|
-
# `claude_review.sh` carries a matching arm ahead of its terminal
|
|
1535
|
-
# arm; relying on its late `*"init"*` catch-all would make routing
|
|
1536
|
-
# depend on `case`-arm order alone.
|
|
1537
|
-
return (
|
|
1538
|
-
"Claude reviewer lane found an unclassifiable host-vocabulary "
|
|
1539
|
-
"entry ("
|
|
1540
|
-
+ ", ".join(
|
|
1541
|
-
safe_identifier(v) for v in sorted(unclassifiable_vocabulary)
|
|
1542
|
-
)
|
|
1543
|
-
+ "); the built-in allowlist cannot prove whether the host or a "
|
|
1544
|
-
"user owns that name, so this lane cannot verify isolation",
|
|
1545
|
-
False,
|
|
1546
|
-
)
|
|
1547
|
-
if (
|
|
1548
|
-
surface_breached
|
|
1549
|
-
or unknown_fields
|
|
1550
|
-
or unverifiable_authority
|
|
1551
|
-
or unclassifiable_vocabulary
|
|
1552
|
-
):
|
|
1021
|
+
if surface_breached or unknown_fields or unverifiable_authority:
|
|
1553
1022
|
if expected_tools:
|
|
1554
1023
|
surface_reason = (
|
|
1555
1024
|
f"{check_label} runtime capability surface does not match the expected boundary; "
|
|
1556
|
-
"a tool
|
|
1025
|
+
"a tool or MCP server is unexpected or unsafe"
|
|
1557
1026
|
)
|
|
1558
1027
|
else:
|
|
1559
1028
|
surface_reason = (
|
|
1560
1029
|
"Claude no-tool probe runtime capability surface is not empty; "
|
|
1561
|
-
"a tool
|
|
1030
|
+
"a tool or MCP server remains declared or invoked"
|
|
1562
1031
|
)
|
|
1563
1032
|
return (
|
|
1564
1033
|
surface_reason,
|
|
@@ -1573,21 +1042,14 @@ def main() -> int:
|
|
|
1573
1042
|
print(
|
|
1574
1043
|
"usage: parse_probe_result.py RC STDOUT_FILE STDERR_FILE "
|
|
1575
1044
|
"[--require-empty-init] [--expected-tools CSV] "
|
|
1576
|
-
"[--expected-native-skills CSV] "
|
|
1577
|
-
"[--required-native-skills CSV] "
|
|
1578
|
-
"[--host-init-baseline FILE] [--validate-host-init-baseline] "
|
|
1579
1045
|
"[--allow-expected-tool-use] [--runtime-surface-only]",
|
|
1580
1046
|
file=sys.stderr,
|
|
1581
1047
|
)
|
|
1582
1048
|
return 64
|
|
1583
1049
|
require_empty_init = False
|
|
1584
1050
|
expected_tools: set[str] = set()
|
|
1585
|
-
expected_native_skills: set[str] = set()
|
|
1586
|
-
required_native_skills: set[str] = set()
|
|
1587
1051
|
allow_expected_tool_use = False
|
|
1588
1052
|
runtime_surface_only = False
|
|
1589
|
-
host_init_baseline_path = ""
|
|
1590
|
-
validate_host_init_baseline = False
|
|
1591
1053
|
option_index = 4
|
|
1592
1054
|
while option_index < len(sys.argv):
|
|
1593
1055
|
option = sys.argv[option_index]
|
|
@@ -1601,32 +1063,12 @@ def main() -> int:
|
|
|
1601
1063
|
if name.strip()
|
|
1602
1064
|
}
|
|
1603
1065
|
option_index += 2
|
|
1604
|
-
elif option == "--expected-native-skills" and option_index + 1 < len(sys.argv):
|
|
1605
|
-
expected_native_skills = {
|
|
1606
|
-
name.strip().lower()
|
|
1607
|
-
for name in sys.argv[option_index + 1].split(",")
|
|
1608
|
-
if name.strip()
|
|
1609
|
-
}
|
|
1610
|
-
option_index += 2
|
|
1611
|
-
elif option == "--required-native-skills" and option_index + 1 < len(sys.argv):
|
|
1612
|
-
required_native_skills = {
|
|
1613
|
-
name.strip().lower()
|
|
1614
|
-
for name in sys.argv[option_index + 1].split(",")
|
|
1615
|
-
if name.strip()
|
|
1616
|
-
}
|
|
1617
|
-
option_index += 2
|
|
1618
1066
|
elif option == "--allow-expected-tool-use":
|
|
1619
1067
|
allow_expected_tool_use = True
|
|
1620
1068
|
option_index += 1
|
|
1621
1069
|
elif option == "--runtime-surface-only":
|
|
1622
1070
|
runtime_surface_only = True
|
|
1623
1071
|
option_index += 1
|
|
1624
|
-
elif option == "--host-init-baseline" and option_index + 1 < len(sys.argv):
|
|
1625
|
-
host_init_baseline_path = sys.argv[option_index + 1]
|
|
1626
|
-
option_index += 2
|
|
1627
|
-
elif option == "--validate-host-init-baseline":
|
|
1628
|
-
validate_host_init_baseline = True
|
|
1629
|
-
option_index += 1
|
|
1630
1072
|
else:
|
|
1631
1073
|
print("unknown or incomplete probe parser option", file=sys.stderr)
|
|
1632
1074
|
return 64
|
|
@@ -1634,13 +1076,7 @@ def main() -> int:
|
|
|
1634
1076
|
# Tool-boundary verdicts are meaningful only with stream-json init
|
|
1635
1077
|
# evidence. Make the stronger requirement intrinsic so a direct caller
|
|
1636
1078
|
# cannot accidentally re-enable the legacy text fallback.
|
|
1637
|
-
if
|
|
1638
|
-
runtime_surface_only
|
|
1639
|
-
or expected_tools
|
|
1640
|
-
or allow_expected_tool_use
|
|
1641
|
-
or expected_native_skills
|
|
1642
|
-
or required_native_skills
|
|
1643
|
-
):
|
|
1079
|
+
if runtime_surface_only or expected_tools or allow_expected_tool_use:
|
|
1644
1080
|
require_empty_init = True
|
|
1645
1081
|
|
|
1646
1082
|
try:
|
|
@@ -1651,20 +1087,6 @@ def main() -> int:
|
|
|
1651
1087
|
|
|
1652
1088
|
stdout = Path(sys.argv[2]).read_text(encoding="utf-8", errors="replace")
|
|
1653
1089
|
stderr = Path(sys.argv[3]).read_text(encoding="utf-8", errors="replace")
|
|
1654
|
-
host_init_baseline = None
|
|
1655
|
-
if validate_host_init_baseline:
|
|
1656
|
-
host_init_baseline, baseline_error = load_host_init_baseline(sys.argv[2])
|
|
1657
|
-
if baseline_error is not None:
|
|
1658
|
-
print(json.dumps({"reason": baseline_error, "generic_failure": False}))
|
|
1659
|
-
return 1
|
|
1660
|
-
return 0
|
|
1661
|
-
if host_init_baseline_path:
|
|
1662
|
-
host_init_baseline, baseline_error = load_host_init_baseline(
|
|
1663
|
-
host_init_baseline_path
|
|
1664
|
-
)
|
|
1665
|
-
if baseline_error is not None:
|
|
1666
|
-
print(json.dumps({"reason": baseline_error, "generic_failure": False}))
|
|
1667
|
-
return 1
|
|
1668
1090
|
normalized_stdout = normalize(stdout)
|
|
1669
1091
|
stderr_surface = classification_surface(None, "", stderr)
|
|
1670
1092
|
env = parse_json_object(normalized_stdout)
|
|
@@ -1714,9 +1136,6 @@ def main() -> int:
|
|
|
1714
1136
|
expected_tools,
|
|
1715
1137
|
allow_expected_tool_use,
|
|
1716
1138
|
runtime_surface_only,
|
|
1717
|
-
expected_native_skills,
|
|
1718
|
-
required_native_skills,
|
|
1719
|
-
host_init_baseline,
|
|
1720
1139
|
):
|
|
1721
1140
|
return 0
|
|
1722
1141
|
elif not require_empty_init:
|
|
@@ -1767,13 +1186,7 @@ def main() -> int:
|
|
|
1767
1186
|
f"{snippet or 'rate limit or quota exceeded'}"
|
|
1768
1187
|
)
|
|
1769
1188
|
else:
|
|
1770
|
-
runtime_state = init_surface_state(
|
|
1771
|
-
events,
|
|
1772
|
-
expected_tools,
|
|
1773
|
-
expected_native_skills,
|
|
1774
|
-
required_native_skills,
|
|
1775
|
-
host_init_baseline,
|
|
1776
|
-
)
|
|
1189
|
+
runtime_state = init_surface_state(events)
|
|
1777
1190
|
runtime_detail = ""
|
|
1778
1191
|
runtime_drift_only = False
|
|
1779
1192
|
if runtime_state is not None:
|
|
@@ -1784,23 +1197,12 @@ def main() -> int:
|
|
|
1784
1197
|
unknown,
|
|
1785
1198
|
unsafe,
|
|
1786
1199
|
unverifiable,
|
|
1787
|
-
vocabulary,
|
|
1788
1200
|
) = runtime_state
|
|
1789
1201
|
# Schema drift routes the same way on the main-invocation path as
|
|
1790
1202
|
# on the probe path. Without this, an unrecognized init container
|
|
1791
1203
|
# would terminate the lane here while merely falling back there —
|
|
1792
1204
|
# the same condition, two verdicts, decided by which parse ran.
|
|
1793
|
-
|
|
1794
|
-
# `unexpected_identifiers` is deliberately absent from the guard,
|
|
1795
|
-
# but NOT because it is redundant with `nonempty` — that was true
|
|
1796
|
-
# only while every unrecognised entry was a breach. An entry it
|
|
1797
|
-
# reports now lands in either `nonempty` (proven customization)
|
|
1798
|
-
# or `vocabulary` (unclassifiable host name), both of which ARE
|
|
1799
|
-
# in the guard; it stays out because it would double-count them
|
|
1800
|
-
# while distinguishing neither. Fixtures pin the combined
|
|
1801
|
-
# drift+breach and vocabulary+breach cases so this stays true
|
|
1802
|
-
# rather than being taken on faith.
|
|
1803
|
-
runtime_drift_only = bool(unknown or unverifiable or vocabulary) and not (
|
|
1205
|
+
runtime_drift_only = bool(unknown or unverifiable) and not (
|
|
1804
1206
|
missing
|
|
1805
1207
|
or nonempty
|
|
1806
1208
|
or unsafe
|
|
@@ -1808,15 +1210,12 @@ def main() -> int:
|
|
|
1808
1210
|
or invoked_tools - expected_tools
|
|
1809
1211
|
or (invoked_tools and not allow_expected_tool_use)
|
|
1810
1212
|
)
|
|
1811
|
-
unexpected_identifiers = unexpected_customization_identifiers(
|
|
1812
|
-
events, expected_native_skills, host_init_baseline
|
|
1813
|
-
)
|
|
1814
1213
|
# EVERY interpolated identifier is CLI-supplied, not just the
|
|
1815
|
-
# field names: tool names
|
|
1816
|
-
#
|
|
1817
|
-
#
|
|
1818
|
-
#
|
|
1819
|
-
#
|
|
1214
|
+
# field names: tool names and invoked-tool names come from the
|
|
1215
|
+
# inspected stream too, and may contain spaces just as freely.
|
|
1216
|
+
# Any one of them left raw lets the inspected CLI pick which
|
|
1217
|
+
# routing arm matches — a breach laundered into "try another
|
|
1218
|
+
# client".
|
|
1820
1219
|
def joined(values):
|
|
1821
1220
|
return ",".join(safe_identifier(v) for v in sorted(values))
|
|
1822
1221
|
|
|
@@ -1825,8 +1224,6 @@ def main() -> int:
|
|
|
1825
1224
|
+ joined(missing)
|
|
1826
1225
|
+ "; unexpected_customizations="
|
|
1827
1226
|
+ joined(nonempty)
|
|
1828
|
-
+ "; unexpected_customization_identifiers="
|
|
1829
|
-
+ joined(unexpected_identifiers)
|
|
1830
1227
|
+ "; declared_tools="
|
|
1831
1228
|
+ joined(declared)
|
|
1832
1229
|
+ "; invoked_tools="
|
|
@@ -1837,22 +1234,8 @@ def main() -> int:
|
|
|
1837
1234
|
+ joined(unsafe)
|
|
1838
1235
|
+ "; unverifiable_authority="
|
|
1839
1236
|
+ joined(unverifiable)
|
|
1840
|
-
+ "; unclassifiable_host_vocabulary="
|
|
1841
|
-
+ joined(vocabulary)
|
|
1842
|
-
)
|
|
1843
|
-
if runtime_drift_only and not (unknown or unverifiable):
|
|
1844
|
-
# The new class gets its own phrase only when it is the SOLE
|
|
1845
|
-
# unverifiable finding. Combined with schema drift or an
|
|
1846
|
-
# unverifiable authority knob, the existing phrase still applies
|
|
1847
|
-
# and still routes to the same fallback-eligible arm, so the
|
|
1848
|
-
# wording of every pre-existing case is left exactly as it was.
|
|
1849
|
-
runtime_reason = (
|
|
1850
|
-
"Claude main invocation found an unclassifiable "
|
|
1851
|
-
"host-vocabulary entry; the built-in allowlist cannot prove "
|
|
1852
|
-
"whether the host or a user owns that name, so this reviewer "
|
|
1853
|
-
"lane cannot verify isolation" + runtime_detail
|
|
1854
1237
|
)
|
|
1855
|
-
|
|
1238
|
+
if runtime_drift_only:
|
|
1856
1239
|
runtime_reason = (
|
|
1857
1240
|
"Claude main invocation found an unrecognized surface-shaped "
|
|
1858
1241
|
"init field; the stream-json schema must be reviewed before "
|
|
@@ -1879,9 +1262,6 @@ def main() -> int:
|
|
|
1879
1262
|
stderr,
|
|
1880
1263
|
expected_tools,
|
|
1881
1264
|
allow_expected_tool_use,
|
|
1882
|
-
expected_native_skills,
|
|
1883
|
-
required_native_skills,
|
|
1884
|
-
host_init_baseline,
|
|
1885
1265
|
)
|
|
1886
1266
|
print(
|
|
1887
1267
|
json.dumps(
|