@ccoalm/ccl-skills 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/SKILL.md +19 -24
  2. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/client-routing.md +32 -32
  3. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/manual-invocation-and-prompts.md +16 -14
  4. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/references/staged-review-contract.md +24 -26
  5. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/AGENTS.md +11 -0
  6. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/claude_review.sh +60 -209
  7. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/init_policy_matrix.py +114 -367
  8. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/parse_probe_result.py +52 -672
  9. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/review_gate.py +10 -2
  10. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/runtime-surface-verification-design.md +4 -2
  11. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_claude_review_probe.sh +77 -444
  12. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_init_policy_matrix.sh +33 -98
  13. package/dist/assets/marketplace/plugins/ccl-skills/skills/code-review/scripts/test_parse_probe_result.sh +57 -173
  14. package/dist/assets/marketplace/plugins/ccl-skills/skills/defect-diagnosis/SKILL.md +1 -1
  15. package/dist/assets/marketplace/plugins/ccl-skills/skills/grill-me/SKILL.md +1 -1
  16. package/dist/assets/marketplace/plugins/ccl-skills/skills/miniapp-product-dev/SKILL.md +1 -1
  17. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-observability/SKILL.md +1 -1
  18. package/dist/assets/marketplace/plugins/ccl-skills/skills/platform-release-engineering/SKILL.md +1 -1
  19. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/SKILL.md +2 -2
  20. package/dist/assets/marketplace/plugins/ccl-skills/skills/product-rd-workflow/references/rd-standards-doc-family-checklist.md +2 -2
  21. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-baseline/SKILL.md +1 -1
  22. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-doc-writer/SKILL.md +1 -1
  23. package/dist/assets/marketplace/plugins/ccl-skills/skills/requirement-scope/SKILL.md +1 -1
  24. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/eval-routing.md +6 -0
  25. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/external-practice-controls.md +1 -1
  26. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/references/source-register.md +39 -0
  27. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/eval-routing-bank.rb +62 -3
  28. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/impact-chain-gate.rb +114 -10
  29. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/review_ledger_binding.py +36 -13
  30. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_impact_chain_refscripts.sh +188 -14
  31. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_check_ccl_regressions.sh +2 -0
  32. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_eval_routing_bank_resolution.sh +253 -0
  33. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_impact_chain_gate_verdict_differential.sh +49 -25
  34. package/dist/assets/marketplace/plugins/ccl-skills/skills/skill-extraction-workflow/scripts/test_review_ledger_binding.sh +63 -5
  35. package/dist/assets/release.json +41 -36
  36. package/package.json +1 -1
@@ -8,8 +8,8 @@ what the parser + the wrapper's routing table actually do end to end.
8
8
  Intended policy (stated here, not derived from the code):
9
9
 
10
10
  A. Isolation proof is the exact `tools` allowlist plus the tool_use scan.
11
- Any declared/invoked tool outside the expectation, or any non-empty
12
- customization list, is a BREACH -> terminal.
11
+ Any declared/invoked tool outside the expectation, or a non-empty MCP
12
+ server list, is a BREACH -> terminal.
13
13
  B. A known field carrying a known-unsafe value (`permissionMode` outside
14
14
  {default, plan}) is a proven BREACH -> terminal.
15
15
  C. Anything we cannot verify -- an unrecognized surface-shaped field, or an
@@ -21,31 +21,23 @@ Intended policy (stated here, not derived from the code):
21
21
  stronger class.
22
22
  F. The verdict must never be steerable by CLI-supplied text: a field NAME
23
23
  cannot select which routing arm matches.
24
- G. HOST VOCABULARY is unverifiable, not a proven breach. The built-in
25
- command/skill allowlists are a snapshot of names the host owns and this
26
- repo does not, so a BARE identifier they do not recognise cannot be shown
27
- to be a user customization -- it is class C, not class A. A NAMESPACED,
28
- path-shaped, duplicated or unparseable entry is still a customization and
29
- stays terminal, and `plugins` is never host vocabulary. Refusal is
30
- unchanged either way; only the next action differs, so no path that was
31
- terminal becomes TOLERATED.
32
- H. A same-executable, no-tool, no-plugin baseline may establish whole-string
33
- host command/skill names only for a formal init reporting the same CLI
34
- version. The baseline never establishes tools, authority, or schema.
35
-
36
- Every row below is (case, init-event, expected verdict class), run through the
37
- parse paths it declares. Policy G rows declare the review-skill paths, because
38
- that is the only invocation shape whose customization lists are populated at all
39
- (measured: 46 commands / 16 skills under `--plugin-dir`, both empty under
40
- `--disable-slash-commands`) -- and it is the shape the skill-free paths never
41
- exercised, which is why an earlier vocabulary outage passed a green matrix.
42
- Each shape is crossed over BOTH parse implementations, since only
43
- `--runtime-surface-only` selects the main-invocation branch.
24
+ G. HOST AND PLUGIN VOCABULARY IS NOT A BOUNDARY. `slash_commands`,
25
+ `terminal_slash_commands`, `skills` and `plugins` are recorded and never
26
+ judged: any value, any shape, any origin, present or absent, is
27
+ TOLERATED on its own. Nothing listed there is invocable past the pinned
28
+ `tools` set, so judging those lists against a snapshot of the host's own
29
+ built-in names turned every CLI release that shipped a new skill or
30
+ command into a reviewer-lane outage while proving nothing. Policies A-E
31
+ still apply unchanged alongside any vocabulary.
32
+
33
+ Every row below is (case, init-event, expected verdict class), run through both
34
+ parse paths: only `--runtime-surface-only` selects the main-invocation branch,
35
+ and the two implementations have drifted apart in opposite directions before.
44
36
 
45
37
  Pass an alternate parser path as argv[1] to check a candidate or a mutant
46
38
  against the same policy -- that is how this oracle is validated: it must report
47
39
  mismatches for a deliberately weakened parser, or its clean verdict means
48
- nothing. That walk is now EXECUTED by `test_init_policy_matrix.sh` rather than
40
+ nothing. That walk is EXECUTED by `test_init_policy_matrix.sh` rather than
49
41
  recorded here as prose: per-mutant scores were a hand-maintained number that
50
42
  every added row invalidated, and a sensitivity claim nothing runs is one
51
43
  refactor away from being vacuous.
@@ -90,26 +82,12 @@ def init(**overrides):
90
82
  return ev
91
83
 
92
84
 
93
- def case(
94
- name,
95
- ev,
96
- expected,
97
- extra_events=(),
98
- paths=("probe", "main"),
99
- host_baseline=None,
100
- ):
101
- """One row. `paths` names the invocation shapes it is meaningful under.
102
-
103
- The two default paths declare no native skills, so their customization
104
- lists are empty in every real run and any entry is a breach. Policy G rows
105
- therefore declare the two review-skill paths instead -- see SKILL_BASE.
106
- """
85
+ def case(name, ev, expected, extra_events=()):
86
+ """One row, run through both parse paths."""
107
87
  return {
108
88
  "name": name,
109
89
  "events": [ev, *extra_events, RESULT],
110
90
  "expected": expected,
111
- "paths": paths,
112
- "host_baseline": host_baseline,
113
91
  }
114
92
 
115
93
 
@@ -123,7 +101,6 @@ ROUTING_PHRASES = [
123
101
  "runtime capability",
124
102
  "Bash tool",
125
103
  "unrecognized surface-shaped init field",
126
- "unclassifiable host-vocabulary entry",
127
104
  ]
128
105
 
129
106
  CASES = [
@@ -175,22 +152,21 @@ CASES = [
175
152
  # --- A: breach -> terminal ---------------------------------------------
176
153
  case("declared-tool", init(tools=["Write"]), TERMINAL),
177
154
  case("declared-bash", init(tools=["Bash"]), TERMINAL),
178
- case("declared-skill", init(skills=["x"]), TERMINAL),
179
- case("declared-plugin", init(plugins=["x"]), TERMINAL),
180
155
  case("declared-mcp", init(mcp_servers=["x"]), TERMINAL),
181
- case("declared-command", init(slash_commands=["x"]), TERMINAL),
156
+ case("declared-mcp-dict", init(mcp_servers=[{"name": "x"}]), TERMINAL),
182
157
  case("invoked-tool", init(), TERMINAL, extra_events=[
183
158
  {"type": "assistant", "message": {"content": [
184
159
  {"type": "tool_use", "name": "Write", "input": {}}]}}]),
185
160
  case("missing-tools", init(tools=...), TERMINAL),
186
161
  case("wrong-type-tools", init(tools="none"), TERMINAL),
187
- case("missing-skills", init(skills=...), TERMINAL),
162
+ case("missing-mcp", init(mcp_servers=...), TERMINAL),
163
+ case("wrong-type-mcp", init(mcp_servers="none"), TERMINAL),
188
164
  case("agents-non-string", init(agents=[{"name": "x"}]), FALLBACK),
189
165
  case("capabilities-non-list", init(capabilities="x"), FALLBACK),
190
166
  case("known-metadata-turned-container", init(cwd={"path": "x"}), FALLBACK),
191
167
 
192
168
  # --- E: stronger class wins on combination ------------------------------
193
- case("drift+breach", init(future_surface=["x"], skills=["y"]), TERMINAL),
169
+ case("drift+breach", init(future_surface=["x"], mcp_servers=["y"]), TERMINAL),
194
170
  case("drift+unsafe-value", init(future_surface=["x"],
195
171
  permissionMode="bypassPermissions"), TERMINAL),
196
172
  case("authority+breach", init(permissionMode=..., tools=["Write"]), TERMINAL),
@@ -217,36 +193,57 @@ for phrase in ROUTING_PHRASES:
217
193
  {**BASE, phrase: ["x"]}, FALLBACK))
218
194
  # the same phrase attached to a genuine breach must NOT soften it
219
195
  CASES.append(case(f"steer-breach[{phrase}]",
220
- {**BASE, "skills": ["y"], phrase: ["x"]}, TERMINAL))
196
+ {**BASE, "mcp_servers": ["y"], phrase: ["x"]}, TERMINAL))
221
197
  # Field names are not the only CLI-supplied text reaching the routed reason:
222
- # tool names, skill/plugin/command identifiers and invoked-tool names are
223
- # interpolated too, and they may contain spaces just as freely. A breach
224
- # must stay terminal no matter what the inspected CLI calls its surfaces.
198
+ # tool names and invoked-tool names are interpolated too, and they may
199
+ # contain spaces just as freely. A breach must stay terminal no matter what
200
+ # the inspected CLI calls its surfaces.
225
201
  CASES.append(case(f"steer-tool-name[{phrase}]",
226
202
  {**BASE, "tools": [phrase]}, TERMINAL))
227
- CASES.append(case(f"steer-skill-name[{phrase}]",
228
- {**BASE, "skills": [phrase]}, TERMINAL))
203
+ CASES.append(case(f"steer-mcp-name[{phrase}]",
204
+ {**BASE, "mcp_servers": [phrase]}, TERMINAL))
229
205
  CASES.append(case(f"steer-invoked-tool[{phrase}]", dict(BASE), TERMINAL,
230
206
  extra_events=[{"type": "assistant", "message": {"content": [
231
207
  {"type": "tool_use", "name": phrase, "input": {}}]}}]))
232
208
 
233
209
 
234
- # --- G: host vocabulary, on the review-skill path --------------------------
235
- # The only shape whose customization lists are populated in a real run. The
236
- # selected skill name must NOT also be a built-in skill name, or the
237
- # ambiguous-selected-owner guard fires on every row and masks the verdict under
238
- # test (observed while measuring the pre-change behaviour).
239
- SELECTED_SKILL = "product-rd-workflow"
240
- SKILL_BASE = {
210
+ # --- G: vocabulary is never a verdict --------------------------------------
211
+ # The owner-aware invocation is the shape whose vocabulary lists are populated
212
+ # in a real run (captured from Claude Code 2.1.261 under `--safe-mode
213
+ # --plugin-dir`); the skill-free shape reports them empty. Both shapes must be
214
+ # accepted with those lists holding anything at all, and every breach class
215
+ # must stay exactly as strong beside them.
216
+ REAL_2_1_261_COMMANDS = [
217
+ "deep-research", "design-sync", "dataviz", "update-config", "verify",
218
+ "debug", "code-review", "simplify", "batch", "fewer-permission-prompts",
219
+ "doctor", "loop", "schedule", "claude-api", "workflow-authoring", "run",
220
+ "run-skill-generator", "advisor", "agents", "auto-mode-setup",
221
+ "autocompact", "clear", "color", "compact", "config", "context", "effort",
222
+ "fast", "heapdump", "init", "mcp", "import", "model", "__remote-workflow",
223
+ "workflow-launch-exec", "reload-plugins", "reload-skills", "rename",
224
+ "ultrareview", "security-review", "usage-credits", "extra-usage", "usage",
225
+ "insights", "recap", "skill-doctor", "goal", "design", "design-consent",
226
+ "design-revoke", "list-agents", "team-onboarding",
227
+ "ccl-skills:product-rd-workflow",
228
+ ]
229
+ REAL_2_1_261_SKILLS = [
230
+ "deep-research", "design-sync", "dataviz", "update-config", "verify",
231
+ "debug", "code-review", "simplify", "batch", "fewer-permission-prompts",
232
+ "doctor", "loop", "schedule", "claude-api", "workflow-authoring", "run",
233
+ "run-skill-generator", "ccl-skills:product-rd-workflow",
234
+ ]
235
+ VOCAB_BASE = {
241
236
  **BASE,
242
- "slash_commands": ["init", "agents"],
243
- "skills": [f"ccl-skills:{SELECTED_SKILL}", "dataviz"],
244
- "plugins": [{"name": "ccl-skills"}],
237
+ "slash_commands": REAL_2_1_261_COMMANDS,
238
+ "terminal_slash_commands": ["doctor", "color", "reload-plugins"],
239
+ "skills": REAL_2_1_261_SKILLS,
240
+ "plugins": [{"name": "ccl-skills", "path": "/p"}],
241
+ "claude_code_version": "2.1.261",
245
242
  }
246
243
 
247
244
 
248
- def skill_init(**overrides):
249
- ev = dict(SKILL_BASE)
245
+ def vocab(**overrides):
246
+ ev = dict(VOCAB_BASE)
250
247
  for key, value in overrides.items():
251
248
  if value is ...:
252
249
  ev.pop(key, None)
@@ -255,293 +252,68 @@ def skill_init(**overrides):
255
252
  return ev
256
253
 
257
254
 
258
- def skill_case(name, ev, expected, extra_events=(), host_baseline=None):
259
- return case(name, ev, expected, extra_events,
260
- paths=("skill", "skill-probe"),
261
- host_baseline=host_baseline)
262
-
263
-
264
255
  def with_command(*extra):
265
- return skill_init(slash_commands=[*SKILL_BASE["slash_commands"], *extra])
256
+ return vocab(slash_commands=[*VOCAB_BASE["slash_commands"], *extra])
266
257
 
267
258
 
268
259
  def with_skill(*extra):
269
- return skill_init(skills=[*SKILL_BASE["skills"], *extra])
260
+ return vocab(skills=[*VOCAB_BASE["skills"], *extra])
270
261
 
271
262
 
272
263
  CASES += [
273
- # the base itself must be accepted, or every row below proves nothing
274
- skill_case("skill-clean", skill_init(), TOLERATED),
275
-
276
- # Real Claude Code 2.1.233 owner-aware init drift. These three commands and
277
- # terminal_slash_commands were captured from the exact safe-mode/plugin
278
- # invocation used by claude_review.sh. They are host vocabulary/metadata,
279
- # not an extra invocable surface; the exact tools list remains independently
280
- # pinned by policy A.
281
- skill_case(
282
- "real-2.1.233-owner",
283
- skill_init(
284
- slash_commands=[
285
- *SKILL_BASE["slash_commands"],
286
- "__remote-workflow",
287
- "auto-mode-setup",
288
- "autocompact",
289
- "list-agents",
290
- ],
291
- terminal_slash_commands=["init", "agents"],
292
- claude_code_version="2.1.233",
293
- ),
294
- TOLERATED,
295
- host_baseline=init(
296
- slash_commands=[
297
- *SKILL_BASE["slash_commands"],
298
- "__remote-workflow",
299
- "auto-mode-setup",
300
- "autocompact",
301
- "list-agents",
302
- ],
303
- skills=["dataviz"],
304
- plugins=[],
305
- terminal_slash_commands=["init", "agents"],
306
- claude_code_version="2.1.233",
307
- ),
308
- ),
309
- skill_case(
310
- "host-baseline-version-mismatch",
311
- skill_init(
312
- slash_commands=[
313
- *SKILL_BASE["slash_commands"],
314
- "auto-mode-setup",
315
- ],
316
- claude_code_version="2.1.234",
317
- ),
318
- FALLBACK,
319
- host_baseline=init(
320
- slash_commands=["auto-mode-setup"],
321
- claude_code_version="2.1.233",
322
- ),
323
- ),
324
- skill_case(
325
- "host-baseline-does-not-allow-unbaselined-command",
326
- skill_init(
327
- slash_commands=[
328
- *SKILL_BASE["slash_commands"],
329
- "baseline-command",
330
- "formal-only-command",
331
- ],
332
- claude_code_version="2.1.233",
333
- ),
334
- FALLBACK,
335
- host_baseline=init(
336
- slash_commands=["baseline-command"],
337
- skills=["dataviz"],
338
- claude_code_version="2.1.233",
339
- ),
340
- ),
341
- skill_case(
342
- "host-baseline-does-not-authorize-new-skill",
343
- skill_init(
344
- skills=[*SKILL_BASE["skills"], "brand-new-host-skill"],
345
- claude_code_version="2.1.233",
346
- ),
347
- FALLBACK,
348
- host_baseline=init(
349
- skills=["brand-new-host-skill"],
350
- claude_code_version="2.1.233",
351
- ),
352
- ),
353
- skill_case(
354
- "host-baseline-rejects-namespaced-command",
355
- skill_init(
356
- slash_commands=[*SKILL_BASE["slash_commands"], "rogue:exfil"],
357
- claude_code_version="2.1.233",
358
- ),
359
- TERMINAL,
360
- host_baseline=init(
361
- slash_commands=["rogue:exfil"],
362
- claude_code_version="2.1.233",
363
- ),
364
- ),
365
- skill_case(
366
- "host-baseline-rejects-namespaced-skill",
367
- skill_init(
368
- skills=[*SKILL_BASE["skills"], "vendor:skill"],
369
- claude_code_version="2.1.233",
370
- ),
371
- TERMINAL,
372
- host_baseline=init(
373
- skills=["vendor:skill"],
374
- claude_code_version="2.1.233",
375
- ),
376
- ),
377
- skill_case(
378
- "host-baseline-does-not-allow-unbaselined-skill",
379
- skill_init(
380
- skills=[*SKILL_BASE["skills"], "formal-only-host-skill"],
381
- claude_code_version="2.1.233",
382
- ),
383
- FALLBACK,
384
- host_baseline=init(
385
- skills=["brand-new-host-skill"],
386
- claude_code_version="2.1.233",
387
- ),
388
- ),
389
- skill_case(
390
- "host-baseline-rejects-required-empty-surface",
391
- skill_init(claude_code_version="2.1.233"),
392
- TERMINAL,
393
- host_baseline=init(
394
- plugins=["untrusted-plugin"],
395
- claude_code_version="2.1.233",
396
- ),
397
- ),
398
- skill_case(
399
- "terminal-commands-must-be-declared",
400
- skill_init(terminal_slash_commands=["not-declared"]),
401
- FALLBACK,
402
- ),
403
- skill_case(
404
- "terminal-commands-ignore-json-key-order",
405
- {
406
- "terminal_slash_commands": ["init", "agents"],
407
- **skill_init(),
408
- },
409
- TOLERATED,
410
- ),
411
- skill_case(
412
- "terminal-commands-must-be-plain-strings",
413
- skill_init(terminal_slash_commands=[{"name": "init"}]),
414
- FALLBACK,
415
- ),
416
-
417
- # the defect: a name the host added and this snapshot does not know
418
- skill_case("host-vocab-new-command", with_command("brand-new-builtin"), FALLBACK),
419
- skill_case("host-vocab-new-skill", with_skill("brand-new-skill"), FALLBACK),
420
- # identifiers are normalized before classification, so case is not a class
421
- skill_case("host-vocab-mixed-case", with_command("BrandNewBuiltin"), FALLBACK),
422
-
423
- # ...and everything that is NOT host vocabulary stays a proven breach
424
- skill_case("namespaced-foreign-command", with_command("evil-plugin:pwn"), TERMINAL),
425
- skill_case("namespaced-foreign-skill", with_skill("evil-plugin:pwn"), TERMINAL),
426
- skill_case("path-shaped-identifier", with_command("dir/cmd"), TERMINAL),
427
- skill_case("unparseable-identifier", with_command("ev!l"), TERMINAL),
428
- skill_case("duplicate-identifiers", with_command("init"), TERMINAL),
429
- skill_case("foreign-plugin",
430
- skill_init(plugins=[{"name": "ccl-skills"}, {"name": "other"}]),
431
- TERMINAL),
432
- # A STRUCTURED entry stays terminal even when its reported `name` is bare:
433
- # the identifier helper reads `name` first, so a sibling key can carry
434
- # path-shaped proof of a real customization that the soft class would then
435
- # ignore. Unread evidence is not the same as absent evidence, which is the
436
- # only thing this class is for. Costs nothing: measured against the real
437
- # CLI, both host-vocabulary fields arrive as plain strings.
438
- skill_case("dict-entry-bare-name",
439
- with_command({"name": "brand-new-builtin", "command": "/x/y"}),
440
- TERMINAL),
441
- # ...and the same shape with no smuggled key is still terminal, so the rule
442
- # is "structured entries are not host vocabulary", not "we grep for paths".
443
- skill_case("dict-entry-bare-name-only",
444
- with_command({"name": "brand-new-builtin"}), TERMINAL),
445
- # The severe variant, and the one a round-5 review found: a structured entry
446
- # whose `name` is an ALLOWED built-in used to clear the allowlist outright,
447
- # so its other keys were never inspected and the run reached TOLERATED with
448
- # isolation reported verified. Reproduced before it was fixed. Both fields,
449
- # because the shape gate must not be per-field folklore.
450
- # The smuggled `name` must be an allowed built-in that is NOT already in the
451
- # base list: reusing one duplicates an identifier, and the duplicate check
452
- # then makes the row terminal for an unrelated reason. Caught by differential
453
- # attribution — with the first fixtures, removing the shape gate flipped
454
- # nothing here, which is a finding about the test, not a clean result.
455
- skill_case("dict-entry-smuggled-under-allowed-command",
456
- with_command({"name": "import", "command": "/x/y",
457
- "extra": ["Bash"]}), TERMINAL),
458
- skill_case("dict-entry-smuggled-under-allowed-skill",
459
- with_skill({"name": "verify", "command": "/x/y"}), TERMINAL),
460
- # ...while `plugins` legitimately carries dicts in every real run, so the
461
- # gate must not spread to it: this is what stops the fix from breaking the
462
- # actual CLI.
463
- skill_case("plugin-dict-stays-legitimate",
464
- skill_init(plugins=[{"name": "ccl-skills", "path": "/p"}]),
465
- TOLERATED),
466
- # The third instance of the same class: a PLAIN STRING whose first token is
467
- # bare while the discarded remainder carries the proof. The identifier helper
468
- # keeps only that first token, so judging the token instead of the whole
469
- # value read `brand-new evil-plugin:pwn` as host vocabulary.
470
- skill_case("whitespace-hides-a-namespace",
471
- with_command("brand-new evil-plugin:pwn"), TERMINAL),
472
- skill_case("whitespace-hides-a-path",
473
- with_command("brand-new dir/cmd"), TERMINAL),
474
- skill_case("whitespace-hides-a-routing-phrase",
475
- with_command("brand-new runtime isolation"), TERMINAL),
476
- skill_case("whitespace-hides-a-namespace-in-skills",
477
- with_skill("brand-new evil-plugin:pwn"), TERMINAL),
478
- # SURROUNDING whitespace is the fourth instance, and the worst: wrapping an
479
- # ALLOWLISTED name reached TOLERATED, because both the allowlist and the
480
- # first version of the whole-value check stripped before comparing.
481
- skill_case("trailing-space-on-an-allowlisted-command",
482
- with_command("import "), TERMINAL),
483
- skill_case("leading-space-on-an-allowlisted-command",
484
- with_command(" import"), TERMINAL),
485
- skill_case("trailing-space-on-an-allowlisted-skill",
486
- with_skill("verify "), TERMINAL),
487
- skill_case("trailing-space-on-an-unknown-command",
488
- with_command("brand-new "), TERMINAL),
489
- skill_case("tab-wrapped-allowlisted-command",
490
- with_command("\timport"), TERMINAL),
491
- # ...and the legitimate namespaced entry must survive all of that, since its
492
- # whole value IS its identifier. Without this row the gate could be tightened
493
- # into rejecting the surface the review-skill mode depends on.
494
- skill_case("selected-namespaced-command-still-allowed",
495
- with_command(f"ccl-skills:{SELECTED_SKILL}"), TOLERATED),
496
-
497
- # E in review-skill mode: the softer class must never absorb a real breach
498
- skill_case("host-vocab+tool-breach",
499
- skill_init(slash_commands=[*SKILL_BASE["slash_commands"], "brand-new"],
500
- tools=["Write"]), TERMINAL),
501
- skill_case("host-vocab+unsafe-value",
502
- skill_init(slash_commands=[*SKILL_BASE["slash_commands"], "brand-new"],
503
- permissionMode="bypassPermissions"), TERMINAL),
504
- skill_case("host-vocab+namespaced",
505
- with_command("brand-new", "evil-plugin:pwn"), TERMINAL),
506
- skill_case("host-vocab+invoked-tool",
507
- with_command("brand-new"), TERMINAL,
508
- extra_events=[{"type": "assistant", "message": {"content": [
509
- {"type": "tool_use", "name": "Write", "input": {}}]}}]),
510
- # two unverifiables are still one unverifiable
511
- skill_case("host-vocab+unknown-container",
512
- skill_init(slash_commands=[*SKILL_BASE["slash_commands"], "brand-new"],
513
- future_surface=["x"]), FALLBACK),
514
-
515
- # per event, not on the union
516
- skill_case("second-init-adds-host-vocab", skill_init(), FALLBACK,
517
- extra_events=[with_command("brand-new-builtin")]),
518
- skill_case("second-init-adds-namespaced", skill_init(), TERMINAL,
519
- extra_events=[with_command("evil-plugin:pwn")]),
520
-
521
- # invariants that must survive in this mode too
522
- skill_case("skill-missing-plugin", skill_init(plugins=[]), TERMINAL),
523
- skill_case("skill-required-absent",
524
- skill_init(skills=["ccl-skills:other-skill"]), TERMINAL),
525
- skill_case("skill-authority-absent", skill_init(permissionMode=...), FALLBACK),
526
- skill_case("skill-declared-tool", skill_init(tools=["Write"]), TERMINAL),
264
+ # the real owner-aware init, and the same on a skill-free run
265
+ case("real-2.1.261-owner", vocab(), TOLERATED),
266
+ case("declared-skill", init(skills=["x"]), TOLERATED),
267
+ case("declared-plugin", init(plugins=["x"]), TOLERATED),
268
+ case("declared-command", init(slash_commands=["x"]), TOLERATED),
269
+ case("missing-skills", init(skills=...), TOLERATED),
270
+ case("missing-commands", init(slash_commands=...), TOLERATED),
271
+ case("missing-plugins", init(plugins=...), TOLERATED),
272
+ # names the host may ship tomorrow, in any spelling
273
+ case("vocab-new-command", with_command("brand-new-builtin"), TOLERATED),
274
+ case("vocab-new-skill", with_skill("brand-new-skill"), TOLERATED),
275
+ case("vocab-mixed-case", with_command("BrandNewBuiltin"), TOLERATED),
276
+ # shapes that used to be read as proof of a customization
277
+ case("vocab-namespaced-command", with_command("other-plugin:cmd"), TOLERATED),
278
+ case("vocab-namespaced-skill", with_skill("other-plugin:skill"), TOLERATED),
279
+ case("vocab-path-shaped", with_command("dir/cmd"), TOLERATED),
280
+ case("vocab-unparseable", with_command("ev!l"), TOLERATED),
281
+ case("vocab-duplicate", with_command("init"), TOLERATED),
282
+ case("vocab-whitespace", with_command(" import", "brand-new dir/cmd"), TOLERATED),
283
+ case("vocab-dict-entry", with_command({"name": "x", "command": "/x/y"}), TOLERATED),
284
+ case("vocab-dict-skill", with_skill({"name": "verify", "command": "/x/y"}), TOLERATED),
285
+ case("vocab-foreign-plugin",
286
+ vocab(plugins=[{"name": "ccl-skills"}, {"name": "other"}]), TOLERATED),
287
+ case("vocab-plugin-strings", vocab(plugins=["ccl-skills"]), TOLERATED),
288
+ case("vocab-no-plugin", vocab(plugins=[]), TOLERATED),
289
+ case("vocab-wrong-type", vocab(skills="none", slash_commands={"a": 1}), TOLERATED),
290
+ case("vocab-terminal-any-shape",
291
+ vocab(terminal_slash_commands=[{"name": "init"}, "not-declared"]), TOLERATED),
292
+ case("vocab-terminal-wrong-type", vocab(terminal_slash_commands="doctor"), TOLERATED),
293
+ case("second-init-adds-vocab", vocab(), TOLERATED,
294
+ extra_events=[with_command("evil-plugin:pwn")]),
295
+
296
+ # ...while every breach class keeps exactly its strength beside vocabulary
297
+ case("vocab+tool-breach", vocab(tools=["Write"]), TERMINAL),
298
+ case("vocab+bash", vocab(tools=["Bash"]), TERMINAL),
299
+ case("vocab+mcp", vocab(mcp_servers=["x"]), TERMINAL),
300
+ case("vocab+unsafe-value", vocab(permissionMode="bypassPermissions"), TERMINAL),
301
+ case("vocab+invoked-tool", vocab(), TERMINAL,
302
+ extra_events=[{"type": "assistant", "message": {"content": [
303
+ {"type": "tool_use", "name": "Write", "input": {}}]}}]),
304
+ case("vocab+unknown-container", vocab(future_surface=["x"]), FALLBACK),
305
+ case("vocab+authority-absent", vocab(permissionMode=...), FALLBACK),
306
+ case("vocab+missing-tools", vocab(tools=...), TERMINAL),
527
307
  ]
528
308
 
529
- # F in review-skill mode: the new class is reached through a CLI-supplied
530
- # IDENTIFIER rather than a field name, so re-run the steering check over it. A
531
- # phrase is normalized to its first token, which is bare -- so the softest arm
532
- # it can reach is its own class, and it must never soften a breach.
309
+ # F over vocabulary: a routing phrase placed INSIDE a vocabulary list is data,
310
+ # never a verdict -- so it is tolerated alone and must not soften a breach.
533
311
  for phrase in ROUTING_PHRASES:
534
- # A phrase containing whitespace cannot be a plain host name at all, so the
535
- # whole-value gate disqualifies it and it stays TERMINAL — stricter than the
536
- # single-token case, and the property under test is unchanged either way: a
537
- # CLI-supplied identifier never reaches an arm SOFTER than its own class.
538
- CASES.append(skill_case(f"steer-vocab-command[{phrase}]",
539
- with_command(phrase),
540
- FALLBACK if " " not in phrase else TERMINAL))
541
- CASES.append(skill_case(f"steer-vocab-breach[{phrase}]",
542
- skill_init(
543
- slash_commands=[*SKILL_BASE["slash_commands"], phrase],
544
- tools=["Write"]), TERMINAL))
312
+ CASES.append(case(f"steer-vocab-command[{phrase}]", with_command(phrase), TOLERATED))
313
+ CASES.append(case(f"steer-vocab-skill[{phrase}]", with_skill(phrase), TOLERATED))
314
+ CASES.append(case(f"steer-vocab-breach[{phrase}]",
315
+ vocab(slash_commands=[*VOCAB_BASE["slash_commands"], phrase],
316
+ tools=["Write"]), TERMINAL))
545
317
 
546
318
 
547
319
  def wrapper_arm(reason: str) -> str:
@@ -573,21 +345,6 @@ PATHS = {
573
345
  "probe": [],
574
346
  "main": ["--require-empty-init", "--expected-tools", "",
575
347
  "--allow-expected-tool-use", "--runtime-surface-only"],
576
- # The review-skill shape, carrying the native-skill flags the wrapper really
577
- # passes. Added because the two paths above declare no skills, so their
578
- # customization lists are empty in every real run -- leaving the branch that
579
- # actually classifies host vocabulary untested by a green matrix.
580
- "skill": ["--require-empty-init", "--expected-tools", "",
581
- "--expected-native-skills", SELECTED_SKILL,
582
- "--required-native-skills", SELECTED_SKILL,
583
- "--allow-expected-tool-use", "--runtime-surface-only"],
584
- # ...and the same shape through the OTHER parse path. `--runtime-surface-only`
585
- # is what selects the main-invocation branch, so without this the review-skill
586
- # cases would only ever exercise one of the two implementations that have
587
- # drifted apart in opposite directions twice before.
588
- "skill-probe": ["--require-empty-init", "--expected-tools", "",
589
- "--expected-native-skills", SELECTED_SKILL,
590
- "--required-native-skills", SELECTED_SKILL],
591
348
  }
592
349
 
593
350
 
@@ -597,18 +354,8 @@ def run_case(entry, parser=PARSER, path="probe"):
597
354
  err = Path(tmp, "stderr")
598
355
  out.write_text("\n".join(json.dumps(ev) for ev in entry["events"]))
599
356
  err.write_text("")
600
- parser_args = [*PATHS[path]]
601
- if entry["host_baseline"] is not None:
602
- baseline = Path(tmp, "host-baseline")
603
- baseline.write_text(
604
- "\n".join(
605
- json.dumps(ev)
606
- for ev in (entry["host_baseline"], RESULT)
607
- )
608
- )
609
- parser_args.extend(["--host-init-baseline", str(baseline)])
610
357
  proc = subprocess.run(
611
- [sys.executable, str(parser), "0", str(out), str(err), *parser_args],
358
+ [sys.executable, str(parser), "0", str(out), str(err), *PATHS[path]],
612
359
  capture_output=True, text=True)
613
360
  if proc.returncode == 0:
614
361
  return TOLERATED, ""
@@ -624,7 +371,7 @@ def main():
624
371
  failures = []
625
372
  runs = 0
626
373
  for entry in CASES:
627
- for path in entry["paths"]:
374
+ for path in PATHS:
628
375
  runs += 1
629
376
  actual, reason = run_case(entry, parser, path)
630
377
  if actual != entry["expected"]: