vigiles 29.0.0 → 30.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/adapter-conformance.d.ts +1 -1
  2. package/dist/adapter-conformance.js +106 -25
  3. package/dist/adapter-registry.d.ts +61 -14
  4. package/dist/adapter-registry.js +78 -10
  5. package/dist/adapter.d.ts +23 -2
  6. package/dist/adapter.js +13 -1
  7. package/dist/adapters/claude-code/adapter.d.ts +32 -2
  8. package/dist/adapters/claude-code/adapter.js +44 -23
  9. package/dist/adapters/claude-code/dialect.js +87 -21
  10. package/dist/adapters/claude-code/hook-protocol.js +16 -0
  11. package/dist/adapters/claude-code/instruction-chain.d.ts +25 -0
  12. package/dist/adapters/claude-code/instruction-chain.js +626 -0
  13. package/dist/adapters/claude-code/layout.d.ts +2 -2
  14. package/dist/adapters/claude-code/layout.js +42 -8
  15. package/dist/adapters/claude-code/model-access.d.ts +41 -0
  16. package/dist/adapters/claude-code/model-access.js +46 -0
  17. package/dist/adapters/claude-code/skill-reachability.d.ts +125 -0
  18. package/dist/adapters/claude-code/skill-reachability.js +111 -0
  19. package/dist/adapters/codex/adapter.d.ts +39 -2
  20. package/dist/adapters/codex/adapter.js +29 -29
  21. package/dist/adapters/codex/dialect.js +11 -6
  22. package/dist/adapters/codex/eval.d.ts +10 -0
  23. package/dist/adapters/codex/eval.js +48 -1
  24. package/dist/adapters/codex/hook-protocol.d.ts +2 -1
  25. package/dist/adapters/codex/hook-protocol.js +10 -0
  26. package/dist/adapters/codex/instruction-chain.d.ts +40 -0
  27. package/dist/adapters/codex/instruction-chain.js +105 -0
  28. package/dist/adapters/codex/layout.d.ts +1 -1
  29. package/dist/adapters/codex/layout.js +41 -14
  30. package/dist/adapters/opencode/adapter.d.ts +33 -2
  31. package/dist/adapters/opencode/adapter.js +36 -36
  32. package/dist/adapters/opencode/dialect.js +2 -2
  33. package/dist/adapters/opencode/instruction-chain.d.ts +37 -0
  34. package/dist/adapters/opencode/instruction-chain.js +70 -0
  35. package/dist/adapters/opencode/layout.d.ts +19 -0
  36. package/dist/adapters/opencode/layout.js +34 -15
  37. package/dist/adoptability.d.ts +31 -1
  38. package/dist/adoptability.js +57 -0
  39. package/dist/cli-main.js +180 -102
  40. package/dist/core/adapter.d.ts +213 -61
  41. package/dist/core/compile.d.ts +2 -2
  42. package/dist/core/compile.js +57 -38
  43. package/dist/core/compose.d.ts +5 -3
  44. package/dist/core/compose.js +5 -3
  45. package/dist/core/config-schema.d.ts +14 -2
  46. package/dist/core/config-schema.js +24 -3
  47. package/dist/core/dialect.d.ts +54 -12
  48. package/dist/core/dialect.js +56 -0
  49. package/dist/core/eval-driver.d.ts +194 -0
  50. package/dist/core/eval-driver.js +3 -0
  51. package/dist/core/frontmatter-read.d.ts +10 -0
  52. package/dist/core/frontmatter-read.js +30 -3
  53. package/dist/core/hook-program.d.ts +27 -2
  54. package/dist/core/hook-program.js +29 -24
  55. package/dist/core/hook-protocol.d.ts +54 -0
  56. package/dist/core/install-reader.d.ts +18 -0
  57. package/dist/core/install-reader.js +88 -0
  58. package/dist/core/instruction-chain.d.ts +444 -0
  59. package/dist/core/instruction-chain.js +292 -0
  60. package/dist/core/instruction-weight.d.ts +96 -14
  61. package/dist/core/instruction-weight.js +65 -30
  62. package/dist/core/layout.d.ts +220 -33
  63. package/dist/core/layout.js +115 -1
  64. package/dist/core/lethal-trifecta.d.ts +12 -7
  65. package/dist/core/lethal-trifecta.js +13 -8
  66. package/dist/core/live-driver.d.ts +137 -0
  67. package/dist/core/live-driver.js +14 -0
  68. package/dist/core/markdown.d.ts +23 -0
  69. package/dist/core/markdown.js +77 -28
  70. package/dist/core/orphans.js +9 -7
  71. package/dist/core/settings-codec.d.ts +17 -0
  72. package/dist/core/settings-codec.js +56 -0
  73. package/dist/core/surface-discovery.d.ts +2 -2
  74. package/dist/core/surface-discovery.js +24 -8
  75. package/dist/core/surface-scopes.d.ts +26 -6
  76. package/dist/core/surface-scopes.js +52 -11
  77. package/dist/core/validate.js +16 -3
  78. package/dist/eval.d.ts +16 -108
  79. package/dist/eval.js +34 -1
  80. package/dist/harness-test.d.ts +3 -63
  81. package/dist/hook-install.d.ts +12 -1
  82. package/dist/hook-install.js +12 -1
  83. package/dist/plugin-loader.d.ts +1 -1
  84. package/dist/plugin-loader.js +43 -36
  85. package/dist/scan-behavioral.d.ts +34 -25
  86. package/dist/scan-behavioral.js +122 -58
  87. package/dist/scan-core.js +37 -18
  88. package/dist/scan-files.d.ts +1 -1
  89. package/dist/scan-files.js +53 -33
  90. package/dist/scan-trigger-suggest.d.ts +0 -21
  91. package/dist/scan-trigger-suggest.js +0 -23
  92. package/dist/scan.d.ts +4 -4
  93. package/dist/scan.js +120 -73
  94. package/dist/skill-harness.d.ts +21 -5
  95. package/dist/skill-harness.js +29 -11
  96. package/dist/surface-discovery-fs.d.ts +2 -0
  97. package/dist/surface-discovery-fs.js +108 -6
  98. package/dist/test-coverage-files.js +24 -17
  99. package/dist/test-coverage.d.ts +9 -3
  100. package/dist/test-coverage.js +32 -17
  101. package/dist/verify-plugin-guards.js +1 -1
  102. package/package.json +1 -1
  103. package/dist/skill-reachability.d.ts +0 -68
  104. package/dist/skill-reachability.js +0 -205
  105. /package/dist/{dialect-drift.d.ts → adapters/claude-code/dialect-drift.d.ts} +0 -0
  106. /package/dist/{dialect-drift.js → adapters/claude-code/dialect-drift.js} +0 -0
package/dist/eval.d.ts CHANGED
@@ -1,4 +1,7 @@
1
- import { parseToolCalls, parseHooks, parseSubagents, type ToolCall, type Trace } from "./harness-test.js";
1
+ import type { HarnessLiveDriver } from "./core/live-driver.js";
2
+ import type { AgentRunArgs, AgentRunner, EvalDriver, EvalUsage, ModelOutputParser, ParsedModelRun, RunOut } from "./core/eval-driver.js";
3
+ export type { AgentRunArgs, AgentRunner, EvalDriver, EvalUsage, ModelOutputParser, ParsedModelRun, RunOut, } from "./core/eval-driver.js";
4
+ import { type ToolCall, type Trace } from "./harness-test.js";
2
5
  import { type CacheMode } from "./eval-cache.js";
3
6
  import { type EvalLockOptions } from "./eval-lock.js";
4
7
  import type { Check, CheckJSON } from "./check.js";
@@ -54,20 +57,6 @@ export interface EvalArm {
54
57
  */
55
58
  readonly effort?: string | number;
56
59
  }
57
- /** Per-run resource use, parsed from the terminal `result` event (0 when absent). */
58
- export interface EvalUsage {
59
- /** `total_cost_usd` reported by claude. */
60
- readonly costUsd: number;
61
- /** Wall-clock `duration_ms` of the run. */
62
- readonly durationMs: number;
63
- /** Fresh (uncached) input tokens, billed at full input price. */
64
- readonly inputTokens: number;
65
- readonly outputTokens: number;
66
- /** Tokens written to the prompt cache this run (~1.25× input price). */
67
- readonly cacheCreationTokens: number;
68
- /** Tokens served from the prompt cache this run (~0.1× input price). */
69
- readonly cacheReadTokens: number;
70
- }
71
60
  /**
72
61
  * Context handed to `measure` after a run, to compute that run's metrics. It is
73
62
  * a `Trace` (so the bare predicates `usedTool` / `skillResolved` / `toolCount` /
@@ -231,52 +220,6 @@ export interface EvalReport {
231
220
  /** True if a `maxCostUsd` budget cap stopped the run before all trials ran. */
232
221
  readonly aborted: boolean;
233
222
  }
234
- /** The raw output of one trial: the agent's exit code + captured streams. */
235
- export interface RunOut {
236
- code: number;
237
- stdout: string;
238
- /** Captured stderr, when the runner provides it (used for rate-limit detection). */
239
- stderr?: string;
240
- }
241
- /** The per-trial arguments handed to an {@link AgentRunner}. */
242
- export interface AgentRunArgs {
243
- readonly task: string;
244
- readonly cwd: string;
245
- readonly model: string;
246
- /**
247
- * Reasoning-budget level for the run (`claude --effort`). Part of the
248
- * MEASUREMENT, not a run knob: it changes the model's output distribution, not
249
- * the sample size — so it lives on the spec next to `model` (never an env),
250
- * and it is hashed into both the cache key and the eval lock. Deliberately
251
- * `string | number` rather than a literal union: the binary accepts an alias
252
- * map, is case-insensitive, and takes an integer budget, and its own valid set
253
- * MOVED between builds (2.1.42 had no `xhigh`, 2.1.257 does) — a hard-coded
254
- * union would reject a valid level after any upstream addition. A wrong value
255
- * is caught at RUNTIME instead, by {@link effortRejection}, which is what the
256
- * binary actually tells us. Omit for the harness default.
257
- */
258
- readonly effort?: string | number;
259
- readonly tools: readonly string[];
260
- readonly hasSettings: boolean;
261
- readonly pluginDir: string | undefined;
262
- readonly timeoutMs: number;
263
- /** Extra env layered over `process.env` for this run (e.g. `VIGILES_INTERCEPT_TOOLS`). */
264
- readonly env?: Record<string, string>;
265
- /**
266
- * When true, `env` is the COMPLETE spawn environment (an ephemeral run env from
267
- * `ephemeralRunEnv`) — the runner does NOT prepend `process.env`, so the
268
- * real `$HOME` / secrets are scrubbed. Default false: `env` is an overlay over
269
- * `process.env` (the byte-identical-to-today path). Set only by `ephemeralEnv`.
270
- */
271
- readonly replaceEnv?: boolean;
272
- }
273
- /**
274
- * Runs one trial and returns its raw output. The default (`spawnAgent`)
275
- * drives the real `claude` CLI; `runEvalWith` takes one explicitly, so the eval
276
- * orchestration is testable without a model (pass a fake returning canned
277
- * stream-json) and a custom runtime can be plugged in.
278
- */
279
- export type AgentRunner = (args: AgentRunArgs) => Promise<RunOut>;
280
223
  /**
281
224
  * Resolve the environment a trial's subprocess actually runs with — the
282
225
  * SECURITY-CRITICAL decision behind `ephemeralEnv`. When `replaceEnv` is set, the
@@ -528,22 +471,6 @@ export declare function checkReportToJUnit(report: CheckReport, opts?: {
528
471
  }): string;
529
472
  /** Parse per-run cost/latency/tokens from a stream — pure, model-free. */
530
473
  export declare function parseUsage(stdout: string): EvalUsage;
531
- /**
532
- * The harness-specific half of a run trace: how a real model's raw stdout maps
533
- * to the common fields. Claude Code's `parseClaudeRun` reads its stream-json; a
534
- * second harness (Codex) supplies its own parser of `codex exec --json` JSONL, so
535
- * the eval tier (`measureTriggerRate`/`runEval`) isn't bound to Claude's format.
536
- * The non-harness fields (cwd/exitCode/stdout/file/sh) stay in `makeContext`.
537
- */
538
- export interface ParsedModelRun {
539
- readonly turns: number;
540
- readonly output: string;
541
- readonly toolCalls: ReturnType<typeof parseToolCalls>;
542
- readonly hooks: ReturnType<typeof parseHooks>;
543
- readonly subagents: ReturnType<typeof parseSubagents>;
544
- readonly usage: EvalUsage;
545
- }
546
- export type ModelOutputParser = (out: RunOut) => ParsedModelRun;
547
474
  /** Parse Claude Code's stream-json stdout into the common trace fields. */
548
475
  export declare function parseClaudeRun(out: RunOut): ParsedModelRun;
549
476
  /**
@@ -864,37 +791,6 @@ export interface TriggerRateReport {
864
791
  */
865
792
  readonly experimental?: string;
866
793
  }
867
- /**
868
- * An eval-tier transport: how to RUN a real harness turn and PARSE its output.
869
- * The default is Claude Code (`claudeEvalDriver`); a second harness supplies its
870
- * own (e.g. `codexEvalDriver` from `vigiles/codex`) and passes it as
871
- * `measureTriggerRate(spec, { evalDriver })` — the eval-tier analog of
872
- * `runHarnessTest`'s `{ adapter }`. `runError` lets the loop drop an
873
- * errored/rate-limited turn instead of scoring it as a miss.
874
- */
875
- export interface EvalDriver {
876
- readonly runner: AgentRunner;
877
- readonly parse: ModelOutputParser;
878
- readonly runError?: (out: RunOut) => string | null;
879
- /**
880
- * The harness this driver runs (e.g. `"claude-code"`, `"codex"`). Folded into a
881
- * trigger-rate eval's LOCK hash so a report recorded on one harness is marked
882
- * STALE if the eval is later switched to another (a different harness can fire a
883
- * skill differently). Optional for back-compat — absent defaults to
884
- * `"claude-code"`, so an existing single-harness lock is unaffected.
885
- */
886
- readonly harness?: string;
887
- /**
888
- * When set, this driver's trigger-rate number is EXPERIMENTAL and not
889
- * validated — the string is the human caveat explaining why (e.g. Codex has no
890
- * skill-selection event, so firing is inferred from a SKILL.md read, which can
891
- * be wrong in both directions). Absent = supported/trustworthy (the default,
892
- * Claude Code). `measureTriggerRate` copies it onto the report and warns; the
893
- * formatter prints it. Precision-first: never let a possibly-wrong number read
894
- * as a measurement.
895
- */
896
- readonly experimental?: string;
897
- }
898
794
  /**
899
795
  * The default (Claude Code) eval driver: real `claude` + stream-json parsing.
900
796
  *
@@ -909,6 +805,18 @@ export interface EvalDriver {
909
805
  * The asymmetry reflects default-vs-injected, not a hexagonal violation.
910
806
  */
911
807
  export declare const claudeEvalDriver: EvalDriver;
808
+ /**
809
+ * The Claude Code {@link HarnessLiveDriver} — the EXECUTING tiers' side of the
810
+ * adapter, reached through `claudeCodeAdapter.liveDriver()`.
811
+ *
812
+ * It lives HERE, at the composition root, for exactly the reason
813
+ * `claudeEvalDriver` above does: it is assembled from the wired default runner
814
+ * and `whichSkillsFired`, and moving it into `src/adapters/claude-code/` would
815
+ * make `eval.ts → adapters/claude-code → eval.ts` a cycle. The adapter reaches
816
+ * it through a dynamic `import()`, so nothing pays for this graph until an
817
+ * executing tier actually runs.
818
+ */
819
+ export declare const claudeCodeLiveDriver: HarnessLiveDriver;
912
820
  /**
913
821
  * Package loose `<skillsDir>/<name>/SKILL.md` skills into a throwaway plugin dir
914
822
  * that `claude --plugin-dir` accepts — so repo-local skills (e.g. `.claude/skills`)
package/dist/eval.js CHANGED
@@ -1,6 +1,6 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
- exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = exports.spawnAgent = exports.EFFORT_ENV_VAR = void 0;
3
+ exports.claudeCodeLiveDriver = exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = exports.spawnAgent = exports.EFFORT_ENV_VAR = void 0;
4
4
  exports.resolveSpawnEnv = resolveSpawnEnv;
5
5
  exports.pinEffortEnv = pinEffortEnv;
6
6
  exports.effortRejection = effortRejection;
@@ -76,6 +76,7 @@ const plugin_loader_js_1 = require("./adapters/claude-code/plugin-loader.js");
76
76
  const runtime_js_1 = require("./adapters/claude-code/runtime.js");
77
77
  const eval_cost_js_1 = require("./eval-cost.js");
78
78
  const proofs_js_1 = require("./core/proofs.js");
79
+ const model_access_js_1 = require("./adapters/claude-code/model-access.js");
79
80
  const harness_test_js_1 = require("./harness-test.js");
80
81
  const eval_cache_js_1 = require("./eval-cache.js");
81
82
  const eval_lock_js_1 = require("./eval-lock.js");
@@ -1413,6 +1414,38 @@ exports.claudeEvalDriver = {
1413
1414
  parse: parseClaudeRun,
1414
1415
  harness: "claude-code",
1415
1416
  };
1417
+ /**
1418
+ * The Claude Code {@link HarnessLiveDriver} — the EXECUTING tiers' side of the
1419
+ * adapter, reached through `claudeCodeAdapter.liveDriver()`.
1420
+ *
1421
+ * It lives HERE, at the composition root, for exactly the reason
1422
+ * `claudeEvalDriver` above does: it is assembled from the wired default runner
1423
+ * and `whichSkillsFired`, and moving it into `src/adapters/claude-code/` would
1424
+ * make `eval.ts → adapters/claude-code → eval.ts` a cycle. The adapter reaches
1425
+ * it through a dynamic `import()`, so nothing pays for this graph until an
1426
+ * executing tier actually runs.
1427
+ */
1428
+ exports.claudeCodeLiveDriver = {
1429
+ evalDriver: exports.claudeEvalDriver,
1430
+ // Env-only, and never a spent token: deciding whether to OFFER a measurement
1431
+ // must not cost one. The three arms are what the consent prompt words
1432
+ // differently — a key bills per token, a session is $0 metered, and neither
1433
+ // present means the tier is skipped with `fix` printed.
1434
+ access: (env) => (0, model_access_js_1.isMeteredAccess)(env)
1435
+ ? { kind: "metered" }
1436
+ : (0, model_access_js_1.hasModelAccess)(env)
1437
+ ? { kind: "subscription" }
1438
+ : { kind: "none", fix: model_access_js_1.CLAUDE_CODE_ACCESS_FIX },
1439
+ // A discrete `Skill` tool_use in the trace says WHICH skill was selected, so
1440
+ // the selection-collision matrix and the adversarial gate can run here.
1441
+ firing: { kind: "event" },
1442
+ // Claude Code namespaces a plugin's skill as `<plugin>:<skill>`; the manifest
1443
+ // name is handed in by the domain, which read it off the layout.
1444
+ firedFor: (skill, plugin) => (t) => whichSkillsFired(t).includes(plugin.name ? `${plugin.name}:${skill}` : skill),
1445
+ // The probe may rebuild the plugin to skills-only stubs: this is the harness
1446
+ // whose plugin shape vigiles packages, so a stubbed rebuild is validated.
1447
+ installsStubs: true,
1448
+ };
1416
1449
  /**
1417
1450
  * Is this directory entry a skill DIRECTORY — following a symlink to one?
1418
1451
  *
@@ -1,5 +1,7 @@
1
1
  import type { HarnessAdapter } from "./core/adapter.js";
2
- import type { HarnessTestDriver, ToolCall, HookFire, ParsedRun, ModelTurn, ModelRequest } from "./core/harness-driver.js";
2
+ import type { HarnessTestDriver, ToolCall, HookFire, ParsedRun, ModelTurn } from "./core/harness-driver.js";
3
+ import type { Trace, SubagentTrace } from "./core/eval-driver.js";
4
+ export type { Trace, SubagentTrace } from "./core/eval-driver.js";
3
5
  import { type SandboxMode } from "./sandbox.js";
4
6
  export { scriptModel } from "./mock-model.js";
5
7
  export type { ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, } from "./core/harness-driver.js";
@@ -56,68 +58,6 @@ export interface HarnessTestSpec {
56
58
  */
57
59
  readonly sandbox?: SandboxMode;
58
60
  }
59
- /**
60
- * The observable record of ONE run — the unified shape produced by BOTH testing
61
- * tiers: `runHarnessTest`'s result and `runEval`'s `measure` ctx (`eval.ts`)
62
- * both satisfy it. That's what lets the bare predicates in `harness-assert.ts`
63
- * (`usedTool` / `skillResolved` / `toolCount` / `toolUsedWith` / `hookFired` /
64
- * `outputContains`) run over either, with the testing helpers asserting and eval
65
- * measuring over the same vocabulary.
66
- */
67
- export interface Trace {
68
- /**
69
- * The tools the agent invoked, each paired with its result — parsed from the
70
- * transcript. Empty unless the run captured the stream (`transcript: true` on
71
- * the harness tier; always on the eval tier). Lets a test assert on the
72
- * agent's *actions* (skills, MCP tools, subagents) instead of grepping stdout.
73
- */
74
- readonly toolCalls: readonly ToolCall[];
75
- /**
76
- * The hooks that fired during the run, each with its decision — parsed from
77
- * the CLI's `hook_response` stream events. Same capture requirement as
78
- * `toolCalls` (empty without the stream). Lets a test assert hook firing
79
- * honestly instead of via a marker file.
80
- */
81
- readonly hooks: readonly HookFire[];
82
- /** The agent's final answer text (the terminal `result` event), or "". */
83
- readonly output: string;
84
- /**
85
- * The requests the model received, captured by the scripted mock — each with
86
- * its `system` prompt and `messages`, flattened to text. Lets a test assert
87
- * what actually reached the model (a SessionStart hook's injected context, a
88
- * slash command's expansion), not just that a hook fired. **Harness tier
89
- * only**: the mock sees the requests, so this is populated by `runHarnessTest`
90
- * (with or without `transcript`); the eval tier drives the real API, so its
91
- * `modelRequests` is always empty.
92
- */
93
- readonly modelRequests: readonly ModelRequest[];
94
- /** Number of model turns. */
95
- readonly turns: number;
96
- /**
97
- * Sub-agent (`Task`) runs as nested traces, keyed by `subagent_type`. A
98
- * subagent runs its own session; CC tags its events with `parent_tool_use_id`
99
- * (= the `Task` tool call) so its tool calls are recovered into a sub-trace
100
- * here, lettng a test assert what the subagent DID (not just that `Task` fired).
101
- * Empty unless the stream was captured / the harness emits subagent events.
102
- */
103
- readonly subagents?: readonly SubagentTrace[];
104
- /** Final contents of a file under the working dir, or null if absent. */
105
- file(path: string): string | null;
106
- }
107
- /** A sub-agent (`Task`) run as a nested trace: its name + the tools it used. */
108
- export interface SubagentTrace {
109
- /** The `subagent_type` from the `Task` tool input. */
110
- readonly name: string;
111
- /** The tools the subagent invoked (events tagged with the Task's id). */
112
- readonly toolCalls: readonly ToolCall[];
113
- /**
114
- * The subagent's RETURNED text — the dispatch tool_result the orchestrator
115
- * receives back. This is where a `result()` contract's `vigiles:ok`/`vigiles:err`
116
- * block lands, so `subagent(name, [output(/vigiles:ok/)])` can assert the typed
117
- * outcome. "" if not captured.
118
- */
119
- readonly output: string;
120
- }
121
61
  export interface HarnessTestResult extends Trace {
122
62
  readonly exitCode: number;
123
63
  readonly stdout: string;
@@ -128,7 +128,18 @@ interface ConfigToml {
128
128
  }
129
129
  /** The TOML sibling of {@link mergeHooksJson} (Codex `[[hooks.<event>]]`). */
130
130
  export declare function mergeHooksToml(existing: ConfigToml, compiled: CompiledHooks, hookPath: string): ConfigToml;
131
- /** Serialize a merged config back to its on-disk text (with trailing newline). */
131
+ /**
132
+ * Serialize a merged config back to its on-disk text.
133
+ *
134
+ * ⚠️ DEPRECATED IN PLACE, not deleted, and the distinction matters: the
135
+ * harness-driven path (`installHookFile` in `cli-main.ts`) goes through
136
+ * `PluginLayout.settings.render` now, so no adapter's encoding is decided here
137
+ * any more. The one remaining caller is `cli-main.ts`'s Codex-plugin wiring,
138
+ * which writes `.codex/config.toml` for a harness it names ITSELF, at the
139
+ * composition root — a caller that already knows the encoding, rather than one
140
+ * branching on a layout field. Its `format` argument is therefore a literal at
141
+ * the call site, not a value read off a port.
142
+ */
132
143
  export declare function serializeConfig(merged: Record<string, unknown>, format: "json" | "toml"): string;
133
144
  export {};
134
145
  //# sourceMappingURL=hook-install.d.ts.map
@@ -298,7 +298,18 @@ function mergeHooksToml(existing, compiled, hookPath) {
298
298
  ]));
299
299
  return { ...existing, hooks: { ...before, ...rewritten } };
300
300
  }
301
- /** Serialize a merged config back to its on-disk text (with trailing newline). */
301
+ /**
302
+ * Serialize a merged config back to its on-disk text.
303
+ *
304
+ * ⚠️ DEPRECATED IN PLACE, not deleted, and the distinction matters: the
305
+ * harness-driven path (`installHookFile` in `cli-main.ts`) goes through
306
+ * `PluginLayout.settings.render` now, so no adapter's encoding is decided here
307
+ * any more. The one remaining caller is `cli-main.ts`'s Codex-plugin wiring,
308
+ * which writes `.codex/config.toml` for a harness it names ITSELF, at the
309
+ * composition root — a caller that already knows the encoding, rather than one
310
+ * branching on a layout field. Its `format` argument is therefore a literal at
311
+ * the call site, not a value read off a port.
312
+ */
302
313
  function serializeConfig(merged, format) {
303
314
  return format === "toml"
304
315
  ? stringifyToml(merged).trimEnd() + "\n"
@@ -1,4 +1,4 @@
1
- import type { PluginLayout } from "./core/layout.js";
1
+ import { type PluginLayout } from "./core/layout.js";
2
2
  import type { ExcludeSet } from "./exclude.js";
3
3
  export interface LoadedPlugin {
4
4
  /** A `.claude/settings.json`-shaped object with hooks resolved. */
@@ -36,8 +36,8 @@ exports.resolveHarness = resolveHarness;
36
36
  */
37
37
  const node_fs_1 = require("node:fs");
38
38
  const node_path_1 = require("node:path");
39
- const toml_1 = require("@iarna/toml");
40
39
  const hash_js_1 = require("./core/hash.js");
40
+ const layout_js_1 = require("./core/layout.js");
41
41
  const exclude_js_1 = require("./exclude.js");
42
42
  const fs_walk_js_1 = require("./fs-walk.js");
43
43
  const source_refs_js_1 = require("./core/source-refs.js");
@@ -57,34 +57,31 @@ function readHooksFile(path) {
57
57
  return safeReadJson(path)?.hooks;
58
58
  }
59
59
  /**
60
- * Parse the layout's manifest in its declared `settingsFormat` — JSON (Claude
61
- * Code's plugin.json) or TOML (Codex's `config.toml`). A TOML harness's manifest
62
- * (hooks, `[mcp_servers]`) would otherwise read as empty through the JSON path.
63
- * Behaviour-identical to `safeReadJson` when the format is JSON.
60
+ * Parse the layout's manifest through its own CODEC. A TOML harness's manifest
61
+ * (hooks, `[mcp_servers]`) would otherwise read as empty through a JSON parse.
62
+ *
63
+ * One code path now, where there used to be `if (settingsFormat === "toml")`
64
+ * over a JSON fallback: the codec is the branch, so a third encoding needs no
65
+ * edit here.
64
66
  */
65
67
  function safeReadManifest(root, layout) {
66
- const path = (0, node_path_1.join)(root, layout.manifestPath);
67
- if (layout.settingsFormat === "toml") {
68
- try {
69
- return (0, toml_1.parse)((0, node_fs_1.readFileSync)(path, "utf-8"));
70
- }
71
- catch {
72
- return null;
73
- }
68
+ try {
69
+ const text = (0, node_fs_1.readFileSync)((0, node_path_1.join)(root, layout.manifestPath), "utf-8");
70
+ return layout.settings.parse(text);
71
+ }
72
+ catch {
73
+ // Missing, unreadable, or malformed — the caller falls through to the
74
+ // other hook locations, exactly as it did for a bad JSON manifest.
75
+ return null;
74
76
  }
75
- return safeReadJson(path);
76
77
  }
77
78
  /**
78
- * Read the `.hooks` field of a settings file in the layout's format — JSON
79
- * (Claude Code's settings.json) or TOML (Codex's `config.toml` `[hooks]`). A
80
- * TOML harness's hooks would otherwise be read as zero by the JSON path.
79
+ * Read the `.hooks` field of a settings file through the layout's codec. A TOML
80
+ * harness's hooks would otherwise be read as zero by a JSON parse.
81
81
  */
82
- function readSettingsHooks(path, format) {
83
- if (format === "json")
84
- return readHooksFile(path);
82
+ function readSettingsHooks(path, codec) {
85
83
  try {
86
- return (0, toml_1.parse)((0, node_fs_1.readFileSync)(path, "utf-8"))
87
- .hooks;
84
+ return codec.parse((0, node_fs_1.readFileSync)(path, "utf-8")).hooks;
88
85
  }
89
86
  catch {
90
87
  return undefined;
@@ -108,12 +105,16 @@ function readHooks(root, layout) {
108
105
  if (m.hooks !== undefined)
109
106
  return m.hooks;
110
107
  }
111
- const conventionPath = (0, node_path_1.join)(root, layout.hooksConventionPath);
112
- if ((0, node_fs_1.existsSync)(conventionPath))
113
- return readHooksFile(conventionPath);
108
+ // Optional: a harness whose hooks are in-process code modules has no
109
+ // standalone hooks FILE, so there is nothing to look for here.
110
+ if (layout.hooksConventionPath !== undefined) {
111
+ const conventionPath = (0, node_path_1.join)(root, layout.hooksConventionPath);
112
+ if ((0, node_fs_1.existsSync)(conventionPath))
113
+ return readHooksFile(conventionPath);
114
+ }
114
115
  const settingsPath = (0, node_path_1.join)(root, layout.settingsPath);
115
116
  if ((0, node_fs_1.existsSync)(settingsPath))
116
- return readSettingsHooks(settingsPath, layout.settingsFormat);
117
+ return readSettingsHooks(settingsPath, layout.settings);
117
118
  return undefined;
118
119
  }
119
120
  /**
@@ -294,7 +295,7 @@ const EMPTY_LOAD = {
294
295
  */
295
296
  function surfaceHasLoadable(layout, surface, tree) {
296
297
  const keys = Object.keys(tree);
297
- return surface === layout.skillDir
298
+ return surface === layout.surfaces.skill
298
299
  ? keys.some((k) => (0, node_path_1.basename)(k) === "SKILL.md")
299
300
  : keys.some((k) => k.endsWith(".md"));
300
301
  }
@@ -317,11 +318,11 @@ function materializeSurfaces(root, layout, files, sources, excluded = exclude_js
317
318
  /** Every surface tree of one scope, read once, keyed by surface dir. */
318
319
  const scopeTrees = (base) => {
319
320
  const trees = new Map();
320
- for (const surface of layout.surfaceDirs)
321
+ for (const surface of (0, layout_js_1.surfaceDirs)(layout))
321
322
  trees.set(surface, surfaceTree((0, node_path_1.join)(root, base, surface)));
322
323
  return trees;
323
324
  };
324
- const hasLoadable = (trees) => layout.surfaceDirs.some((s) => surfaceHasLoadable(layout, s, trees.get(s) ?? {}));
325
+ const hasLoadable = (trees) => (0, layout_js_1.surfaceDirs)(layout).some((s) => surfaceHasLoadable(layout, s, trees.get(s) ?? {}));
325
326
  const add = (key, content, onDisk) => {
326
327
  files[key] = content;
327
328
  sources[key] = onDisk;
@@ -342,7 +343,7 @@ function materializeSurfaces(root, layout, files, sources, excluded = exclude_js
342
343
  declaredTrees.set(base, scopeTrees(base));
343
344
  /** Copy one scope's already-read trees into `files`, keyed by that scope. */
344
345
  const materializeScope = (scope, trees) => {
345
- for (const surface of layout.surfaceDirs) {
346
+ for (const surface of (0, layout_js_1.surfaceDirs)(layout)) {
346
347
  const tree = trees.get(surface) ?? {};
347
348
  for (const [rel, content] of Object.entries(tree))
348
349
  add((0, surface_scopes_js_1.scopeKey)(scope, surface, rel), content, (0, node_path_1.join)(root, scope.base, surface, rel));
@@ -397,7 +398,8 @@ function materializeSurfaces(root, layout, files, sources, excluded = exclude_js
397
398
  skillName: (0, node_path_1.basename)(root),
398
399
  rootHasLoadable: hasLoadable(rootTrees),
399
400
  isPluginShaped: (0, node_fs_1.existsSync)((0, node_path_1.join)(root, layout.manifestPath)) ||
400
- (0, node_fs_1.existsSync)((0, node_path_1.join)(root, layout.hooksConventionPath)),
401
+ (layout.hooksConventionPath !== undefined &&
402
+ (0, node_fs_1.existsSync)((0, node_path_1.join)(root, layout.hooksConventionPath))),
401
403
  userHasLoadable: hasLoadable(userTrees),
402
404
  declaredRoots,
403
405
  });
@@ -409,11 +411,16 @@ function materializeSurfaces(root, layout, files, sources, excluded = exclude_js
409
411
  // named (`vigiles audit <dir>`), not one this walk discovered, and refusing
410
412
  // to read the path someone explicitly pointed at is not a containment rule.
411
413
  const tree = readTree(root, root, excluded);
414
+ // The skill dir is CARRIED on the variant, because only a layout that has
415
+ // one can produce it. This used to re-read `layout.surfaces.skill` and
416
+ // guard the `undefined` case with a `return` that nothing could reach —
417
+ // see `SurfaceSource` for why that shape kept coming back.
418
+ const { skillDir } = source;
412
419
  for (const [rel, content] of Object.entries(tree)) {
413
- add((0, node_path_1.join)(layout.materializeRoot, layout.skillDir, source.skillName, rel), content, (0, node_path_1.join)(root, rel));
420
+ add((0, node_path_1.join)((0, layout_js_1.materializePrefix)(layout), skillDir, source.skillName, rel), content, (0, node_path_1.join)(root, rel));
414
421
  }
415
- counts[layout.skillDir] = Object.keys(tree).length;
416
- harnessCounts[layout.skillDir] = counts[layout.skillDir];
422
+ counts[skillDir] = Object.keys(tree).length;
423
+ harnessCounts[skillDir] = counts[skillDir];
417
424
  return { counts, harnessCounts, scopes: [] };
418
425
  }
419
426
  case "scopes": {
@@ -474,7 +481,7 @@ function hasMcp(root, layout) {
474
481
  // so this and its browser twin (scan-files.ts) cannot disagree on them and
475
482
  // neither can omit one. See that module for the two boundary defects.
476
483
  function intraRefRe(layout) {
477
- return (0, source_refs_js_1.intraRefPattern)(layout.intraRefDirs);
484
+ return (0, source_refs_js_1.intraRefPattern)((0, layout_js_1.executableSourceDirs)(layout));
478
485
  }
479
486
  // Shell vars that root a path OUTSIDE the plugin (the user's project / home), so
480
487
  // a `surface/…` after one is NOT a plugin-root ref. Anything else ($ROOT,
@@ -572,7 +579,7 @@ function missingRefsIn(content, re, root) {
572
579
  /** The plugin's executable (non-prose) source files under the surface dirs. */
573
580
  function executableSources(root, layout) {
574
581
  const sources = {};
575
- for (const surface of layout.intraRefDirs) {
582
+ for (const surface of (0, layout_js_1.executableSourceDirs)(layout)) {
576
583
  const dir = (0, node_path_1.join)(root, surface);
577
584
  if (!(0, node_fs_1.existsSync)(dir) || !(0, node_fs_1.statSync)(dir).isDirectory())
578
585
  continue;
@@ -15,9 +15,8 @@
15
15
  import type { PluginLayout } from "./core/layout.js";
16
16
  import type { HarnessDialect } from "./core/dialect.js";
17
17
  import { type EvalDriver } from "./eval.js";
18
- import { type Trace } from "./harness-test.js";
19
- /** Which harness drives the behavioral column (default Claude Code). */
20
- export type ProbeHarness = "claude-code" | "codex";
18
+ import { type EventFiringDriver, type HarnessLiveDriver } from "./core/live-driver.js";
19
+ import type { HarnessAdapter } from "./core/adapter.js";
21
20
  /** Author-supplied prompt sets for one skill (bare skill name → these). */
22
21
  export interface SkillPrompts {
23
22
  readonly prompts: readonly string[];
@@ -53,28 +52,16 @@ export interface ProbeOptions {
53
52
  readonly model?: string;
54
53
  readonly minPrompts?: number;
55
54
  readonly minDistance?: number;
56
- /** Which harness to drive (default `"claude-code"`). */
57
- readonly harness?: ProbeHarness;
55
+ /** Which harness drives it — the adapter itself, not its name (default
56
+ * {@link defaultAdapter}). A reference-only adapter reports `available: false`. */
57
+ readonly adapter?: HarnessAdapter;
58
58
  /** Layout + dialect for candidate discovery — so a Codex repo's skills (under
59
59
  * the Codex layout) are found, not silently missed by the default CC layout. */
60
60
  readonly layout?: PluginLayout;
61
61
  readonly dialect?: HarnessDialect;
62
62
  }
63
- /**
64
- * Per-harness probe wiring: the eval driver (runner+parse), how to build the
65
- * `fired` predicate for a skill, whether to stub bodies, and an availability
66
- * gate. Claude detects firing via the `Skill` tool_use (namespaced by the
67
- * plugin name); Codex has no skill event, so firing is the SKILL.md read
68
- * (`codexSkillFired`, bare name) — see `research/codex-prototype-findings.md`.
69
- */
70
- export interface HarnessProbe {
71
- readonly evalDriver: EvalDriver;
72
- readonly firedFor: (name: string) => (t: Trace) => boolean;
73
- readonly stub: boolean;
74
- readonly available: () => boolean;
75
- }
76
63
  /** The injectable core (for tests): probe every model-invocable skill that has prompts. */
77
- export declare function probePluginTriggersWith(dir: string, promptSet: TriggerPromptSet, probe: HarnessProbe, opts?: ProbeOptions): Promise<BehavioralReport>;
64
+ export declare function probePluginTriggersWith(dir: string, promptSet: TriggerPromptSet, live: HarnessLiveDriver, opts?: ProbeOptions): Promise<BehavioralReport>;
78
65
  /**
79
66
  * Probe a plugin's skills against the real harness (default Claude Code; Codex via
80
67
  * `opts.harness`). Needs that harness's binary + auth; degrades to
@@ -101,8 +88,21 @@ export interface SelectionOptions {
101
88
  readonly effort?: string | number;
102
89
  /** Parallel runs across the prompts × trials grid (default 1). */
103
90
  readonly concurrency?: number;
104
- /** Which harness drives it (default `"claude-code"`; others report n/a). */
105
- readonly harness?: ProbeHarness;
91
+ /** Which harness drives it — the adapter itself (default {@link defaultAdapter}).
92
+ * A harness whose firing signal is INFERRED rather than a discrete event reports
93
+ * n/a: the matrix asks WHICH skill fired, which an inference cannot answer. */
94
+ readonly adapter?: HarnessAdapter;
95
+ /**
96
+ * @deprecated Pass {@link SelectionOptions.adapter} instead. Kept for one
97
+ * release because this options type is public API (`vigiles/claude-code`);
98
+ * a name is resolved through the registry, never compared to a literal.
99
+ */
100
+ readonly harness?: string;
101
+ /** Layout + dialect for candidate discovery, mirroring {@link ProbeOptions} —
102
+ * so a Codex repo's skills are found under the Codex layout, not silently
103
+ * missed by the Claude Code one. */
104
+ readonly layout?: PluginLayout;
105
+ readonly dialect?: HarnessDialect;
106
106
  }
107
107
  /** One run's outcome for the matrix: which of the plugin's OWN skills fired. */
108
108
  interface SelectionRun {
@@ -142,8 +142,16 @@ export interface SelectionReport {
142
142
  * model-driving so it's unit-testable with synthetic runs (no model).
143
143
  */
144
144
  export declare function buildSelectionReport(skills: readonly string[], runs: readonly SelectionRun[]): SelectionReport;
145
- /** The injectable core (for tests): drive the matrix via a fake/real probe. */
146
- export declare function measurePluginSelectionWith(dir: string, promptSet: TriggerPromptSet, probe: HarnessProbe, opts?: SelectionOptions): Promise<SelectionReport>;
145
+ /**
146
+ * The injectable core (for tests): drive the matrix via a fake/real driver.
147
+ *
148
+ * 🔴 IT TAKES AN {@link EventFiringDriver}, NOT ANY LIVE DRIVER, AND THAT IS THE
149
+ * RATCHET. The matrix asks WHICH of the plugin's skills fired on each prompt; a
150
+ * harness that only INFERS firing cannot answer that, and the old guard was a
151
+ * name check in the wrapper below — invisible to anyone calling this core
152
+ * directly. Now an inferred driver is a compile error at every call site.
153
+ */
154
+ export declare function measurePluginSelectionWith(dir: string, promptSet: TriggerPromptSet, probe: EventFiringDriver, opts?: SelectionOptions): Promise<SelectionReport>;
147
155
  /**
148
156
  * Measure a plugin's cross-skill selection-collision matrix against the real
149
157
  * harness (Claude Code only — Codex has no skill-selection event). Needs the
@@ -181,7 +189,7 @@ export declare function measureSelectionMatrix(dir: string, opts?: SelectionMatr
181
189
  * Injectable core of {@link measureSelectionMatrix} (for tests): auto-derive the
182
190
  * prompts (unless supplied) and drive the matrix via a fake/real probe.
183
191
  */
184
- export declare function measureSelectionMatrixWith(dir: string, probe: HarnessProbe, opts?: SelectionMatrixOptions): Promise<SelectionReport>;
192
+ export declare function measureSelectionMatrixWith(dir: string, probe: EventFiringDriver, opts?: SelectionMatrixOptions): Promise<SelectionReport>;
185
193
  /**
186
194
  * Assert a plugin's skills don't hijack each other — the gate over a
187
195
  * {@link SelectionReport} from {@link measureSelectionMatrix}. `maxOffDiagonal`
@@ -239,7 +247,8 @@ export interface GateOptions {
239
247
  readonly trials?: number;
240
248
  /** Concurrent harness runs (default 1). */
241
249
  readonly concurrency?: number;
242
- readonly harness?: ProbeHarness;
250
+ /** Which harness drives it — the adapter itself (default {@link defaultAdapter}). */
251
+ readonly adapter?: HarnessAdapter;
243
252
  readonly layout?: PluginLayout;
244
253
  readonly dialect?: HarnessDialect;
245
254
  /** Author-supplied attack prompts (bare skill name → prompts); overrides derive. */