vigiles 25.0.0 → 26.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/adapters/claude-code/run-scripts.d.ts +5 -2
- package/dist/adapters/claude-code/run-scripts.js +6 -3
- package/dist/adapters/codex/eval.d.ts +18 -0
- package/dist/adapters/codex/eval.js +27 -0
- package/dist/cli.d.ts +2 -1
- package/dist/cli.js +157 -74
- package/dist/core/adopt.d.ts +50 -1
- package/dist/core/adopt.js +100 -6
- package/dist/core/bash-effects.d.ts +11 -0
- package/dist/core/bash-effects.js +52 -12
- package/dist/core/bash-equivalents.d.ts +18 -0
- package/dist/core/bash-equivalents.js +239 -0
- package/dist/core/coverage.d.ts +3 -3
- package/dist/core/coverage.js +10 -6
- package/dist/core/hook-program.js +48 -2
- package/dist/core/orphans.d.ts +8 -1
- package/dist/core/orphans.js +8 -2
- package/dist/core/types.d.ts +14 -6
- package/dist/eval-cache.d.ts +7 -0
- package/dist/eval-lock.d.ts +20 -0
- package/dist/eval.d.ts +117 -4
- package/dist/eval.js +164 -32
- package/dist/exclude.d.ts +25 -0
- package/dist/exclude.js +132 -0
- package/dist/guardrail-check.d.ts +43 -24
- package/dist/guardrail-check.js +57 -0
- package/dist/scan-behavioral.d.ts +10 -0
- package/dist/scan-behavioral.js +1 -0
- package/dist/test.d.ts +1 -1
- package/dist/test.js +3 -2
- package/package.json +1 -1
package/dist/core/coverage.js
CHANGED
|
@@ -36,7 +36,7 @@ function readNpmScripts(basePath) {
|
|
|
36
36
|
* Collect commands documented in specs by loading spec source files directly.
|
|
37
37
|
* Reads the structured `commands` field — no markdown parsing.
|
|
38
38
|
*/
|
|
39
|
-
function collectDocumentedCommands(basePath, specs) {
|
|
39
|
+
function collectDocumentedCommands(basePath, specs, ignore) {
|
|
40
40
|
const commands = new Set();
|
|
41
41
|
if (specs) {
|
|
42
42
|
for (const spec of specs) {
|
|
@@ -50,8 +50,12 @@ function collectDocumentedCommands(basePath, specs) {
|
|
|
50
50
|
// Fallback: scan compiled markdown for spec file references, then
|
|
51
51
|
// load the compiled JS spec from dist/. If that fails, try to
|
|
52
52
|
// extract commands from the compiled output (last resort).
|
|
53
|
+
// `ignore` is the repo's ExcludeSet string face (src/exclude.ts): the floor
|
|
54
|
+
// plus `.vigilesrc.json#exclude`, required so this fallback cannot walk a
|
|
55
|
+
// vendored corpus the repo excluded (#192). The CLI always passes `specs`,
|
|
56
|
+
// so this branch is reached by the library path and tests only.
|
|
53
57
|
const mdFiles = (0, glob_1.globSync)("**/*.md", {
|
|
54
|
-
ignore: [
|
|
58
|
+
ignore: [...ignore],
|
|
55
59
|
cwd: basePath,
|
|
56
60
|
});
|
|
57
61
|
for (const mdFile of mdFiles) {
|
|
@@ -92,9 +96,9 @@ function collectDocumentedCommands(basePath, specs) {
|
|
|
92
96
|
/**
|
|
93
97
|
* Compute script coverage: what % of npm scripts are documented in specs.
|
|
94
98
|
*/
|
|
95
|
-
function computeScriptCoverage(basePath, threshold, specs) {
|
|
99
|
+
function computeScriptCoverage(basePath, threshold, specs, ignore) {
|
|
96
100
|
const allScripts = readNpmScripts(basePath);
|
|
97
|
-
const documented = collectDocumentedCommands(basePath, specs);
|
|
101
|
+
const documented = collectDocumentedCommands(basePath, specs, ignore);
|
|
98
102
|
const covered = [];
|
|
99
103
|
const uncovered = [];
|
|
100
104
|
for (const script of allScripts) {
|
|
@@ -146,13 +150,13 @@ function computeLinterRuleCoverage(enabled, documented, threshold) {
|
|
|
146
150
|
/**
|
|
147
151
|
* Check all coverage metrics against thresholds.
|
|
148
152
|
*/
|
|
149
|
-
function checkCoverage(basePath, thresholds, linterEnabled, linterDocumented, specs) {
|
|
153
|
+
function checkCoverage(basePath, thresholds, linterEnabled, linterDocumented, specs, ignore) {
|
|
150
154
|
const metrics = [];
|
|
151
155
|
// Linter rule coverage
|
|
152
156
|
const linterMetric = computeLinterRuleCoverage(linterEnabled, linterDocumented, thresholds.linterRules);
|
|
153
157
|
metrics.push(linterMetric);
|
|
154
158
|
// Script coverage
|
|
155
|
-
const scriptMetric = computeScriptCoverage(basePath, thresholds.scripts, specs);
|
|
159
|
+
const scriptMetric = computeScriptCoverage(basePath, thresholds.scripts, specs, ignore);
|
|
156
160
|
metrics.push(scriptMetric);
|
|
157
161
|
const passing = metrics.every((m) => m.passing);
|
|
158
162
|
return { metrics, passing };
|
|
@@ -489,7 +489,48 @@ function commandView(raw, root) {
|
|
|
489
489
|
raw,
|
|
490
490
|
runs(program, opts) {
|
|
491
491
|
const tokens = program.split(/\s+/).filter(Boolean);
|
|
492
|
-
|
|
492
|
+
// 🔴 READS RAW **AND** NORMALIZED ARGV — the same union `touches` already
|
|
493
|
+
// uses, and for the same reason. This matcher read only `leaves` (raw), so
|
|
494
|
+
// three ordinary rewrites walked past a guard that blocks the plain form.
|
|
495
|
+
// MEASURED 2026-09-02 on the shipped dogfood artifact behind the public
|
|
496
|
+
// 7/7 claim (`examples/harness/safe-bash-guard.mjs`):
|
|
497
|
+
// git push --force origin main exit 2 blocked
|
|
498
|
+
// git push "--force" origin main exit 0 ALLOWED
|
|
499
|
+
// sudo git push --force origin main exit 0 ALLOWED
|
|
500
|
+
// /usr/bin/git push --force origin main exit 0 ALLOWED
|
|
501
|
+
// Across 30 shell-equivalent variants of the seven catalog seeds the guard
|
|
502
|
+
// blocked 8. The seeds were 7/7 — the number was true and measured on the
|
|
503
|
+
// only forms anyone had written down.
|
|
504
|
+
//
|
|
505
|
+
// The quoting case is the sharpest: `getLiteral` (bash-effects) returns
|
|
506
|
+
// null for a quoted word, and `leafCommands` FILTERS nulls, so
|
|
507
|
+
// `git push "--force" origin main` arrives as [git, push, origin, main] —
|
|
508
|
+
// a NON-force push. The force check was answering correctly about an argv
|
|
509
|
+
// that had already lost the flag.
|
|
510
|
+
//
|
|
511
|
+
// Force is asked of the NORMALIZED leaf via `hasFlag`, not by re-scanning
|
|
512
|
+
// strings: short clusters are already split (`-rf`→r,f) and each flag is
|
|
513
|
+
// recorded in both short and long form, so `-f` and `--force` are one
|
|
514
|
+
// question. The raw half stays because it is what an exact-literal match
|
|
515
|
+
// is for; neither half alone is enough, which is exactly what the comment
|
|
516
|
+
// on `touches` says about its own union.
|
|
517
|
+
const rawHit = leaves.some((argv) => runsSeq(argv, tokens) && (opts?.force ? hasForce(argv) : true));
|
|
518
|
+
if (rawHit)
|
|
519
|
+
return true;
|
|
520
|
+
// A FLAG named in the pattern is asked of `hasFlag`, not matched as a
|
|
521
|
+
// literal token. The normalized leaf already records every flag in BOTH
|
|
522
|
+
// short and long form, so `runs("git commit --no-verify")` also catches
|
|
523
|
+
// `git commit -n` — which the literal path cannot, because `-n` is simply
|
|
524
|
+
// a different string. Found by the generated battery on its first real
|
|
525
|
+
// run: 72/73, and the one miss was exactly this
|
|
526
|
+
// (`git commit -n -m 'skip hooks'`). The `{force:true}` option exists
|
|
527
|
+
// because force needed this for `-f`; every other flag needed it too and
|
|
528
|
+
// nobody noticed, since nobody wrote the short form down by hand.
|
|
529
|
+
const flagTokens = tokens.filter((t) => t.startsWith("-"));
|
|
530
|
+
const plainTokens = tokens.filter((t) => !t.startsWith("-"));
|
|
531
|
+
return normalized.some((leaf) => runsSeq(leaf.argv, plainTokens) &&
|
|
532
|
+
flagTokens.every((t) => leaf.hasFlag(t.replace(/^-+/, ""))) &&
|
|
533
|
+
(opts?.force ? leaf.hasFlag("force", "f") : true));
|
|
493
534
|
},
|
|
494
535
|
isSideEffecting: () => (0, bash_effects_js_1.classifyBashCommand)(raw) === "side-effecting",
|
|
495
536
|
// Both of these are DENYLIST matchers, so both break an undecidable verdict
|
|
@@ -532,7 +573,12 @@ function commandView(raw, root) {
|
|
|
532
573
|
// normalized ones, and a gate built on both had a silent hole).
|
|
533
574
|
writesTo: (prefixes) => matchedWriteTargets(prefixes).length > 0,
|
|
534
575
|
writeTargets: matchedWriteTargets,
|
|
535
|
-
|
|
576
|
+
// Same union, same reason: `curl … | /bin/sh` and `curl … | sudo sh` are a
|
|
577
|
+
// pipe-to-shell, and a raw-only check misses both (`isBareShellLeaf` tests
|
|
578
|
+
// the head literally, and the normalizer is what reduces `/bin/sh`→`sh` and
|
|
579
|
+
// strips the wrapper).
|
|
580
|
+
pipesToShell: () => leaves.some(isBareShellLeaf) ||
|
|
581
|
+
normalized.some((leaf) => isBareShellLeaf(leaf.argv)),
|
|
536
582
|
};
|
|
537
583
|
}
|
|
538
584
|
const tool = (name) => ({ tool: name });
|
package/dist/core/orphans.d.ts
CHANGED
|
@@ -32,8 +32,15 @@ export interface FindOrphansOptions {
|
|
|
32
32
|
* or to your project's doc globs (e.g. `["wiki/**\/*.md"]`) to override.
|
|
33
33
|
*/
|
|
34
34
|
readonly include?: readonly string[];
|
|
35
|
-
/** Glob patterns to exclude within the include scope. */
|
|
35
|
+
/** Glob patterns to exclude within the include scope (orphan CANDIDACY only). */
|
|
36
36
|
readonly exclude?: readonly string[];
|
|
37
|
+
/**
|
|
38
|
+
* The repo-wide `.vigilesrc.json#exclude` (the ExcludeSet string face,
|
|
39
|
+
* src/exclude.ts). Applied to BOTH walks — candidates AND the reference scan —
|
|
40
|
+
* so an excluded corpus can neither be an orphan nor keep one alive (#192).
|
|
41
|
+
* The CLI always passes it; a direct library caller may omit it.
|
|
42
|
+
*/
|
|
43
|
+
readonly repoExclude?: readonly string[];
|
|
37
44
|
/**
|
|
38
45
|
* Harnesses whose surface files (instruction file, `SKILL.md`, subagents,
|
|
39
46
|
* commands) are load-bearing by location and thus never orphan CANDIDATES
|
package/dist/core/orphans.js
CHANGED
|
@@ -153,12 +153,18 @@ function findOrphanDocs(options = {}) {
|
|
|
153
153
|
const basePath = options.basePath ?? process.cwd();
|
|
154
154
|
const include = options.include ?? DEFAULT_INCLUDE;
|
|
155
155
|
const userExclude = options.exclude ?? [];
|
|
156
|
-
|
|
156
|
+
// Two exclusions with two jobs (#192). `repoExclude` is the repo-wide
|
|
157
|
+
// `.vigilesrc.json#exclude` — a vendored corpus is neither an orphan
|
|
158
|
+
// CANDIDATE nor a SOURCE of references (a link from inside it must not keep a
|
|
159
|
+
// doc alive). The rule's own `exclude` only narrows candidacy: a doc kept out
|
|
160
|
+
// of the orphan list can still reference others. Union, never override.
|
|
161
|
+
const repoExclude = options.repoExclude ?? [];
|
|
162
|
+
const ignore = [...DEFAULT_IGNORE, ...repoExclude, ...userExclude];
|
|
157
163
|
const layouts = options.layouts ?? [];
|
|
158
164
|
const allDocs = collectDocs(basePath, include, ignore, layouts);
|
|
159
165
|
const allMarkdown = (0, glob_1.globSync)("**/*.md", {
|
|
160
166
|
cwd: basePath,
|
|
161
|
-
ignore: [...DEFAULT_IGNORE],
|
|
167
|
+
ignore: [...DEFAULT_IGNORE, ...repoExclude],
|
|
162
168
|
});
|
|
163
169
|
const referencedBy = new Map();
|
|
164
170
|
for (const mdPath of allMarkdown) {
|
package/dist/core/types.d.ts
CHANGED
|
@@ -409,12 +409,20 @@ export interface VigilesConfig {
|
|
|
409
409
|
bundles?: "root" | "all";
|
|
410
410
|
orphans?: OrphansConfig;
|
|
411
411
|
/**
|
|
412
|
-
*
|
|
413
|
-
* (tsconfig-style, relative to the
|
|
414
|
-
*
|
|
415
|
-
* `
|
|
416
|
-
*
|
|
417
|
-
*
|
|
412
|
+
* Paths and globs the repo's own tooling does NOT police — vendored corpora,
|
|
413
|
+
* benchmark fixtures, frozen reproductions (tsconfig-style, relative to the
|
|
414
|
+
* repo root; a bare directory name such as `"bench"` excludes its subtree, as
|
|
415
|
+
* do `"bench/"` and `"bench/**"`). `node_modules`/`dist`/`.git`/`.vigiles` are
|
|
416
|
+
* always excluded.
|
|
417
|
+
*
|
|
418
|
+
* ONE filter, every pass (#192): `compile` does not load an excluded spec,
|
|
419
|
+
* `lint` does not discover an excluded instruction file, nested bundle, doc, or
|
|
420
|
+
* surface, `audit` does not read an excluded instruction file, and
|
|
421
|
+
* `test`/`eval` do not discover an excluded script. It filters DISCOVERY only:
|
|
422
|
+
* a path you name on the command line is still processed, and one line says
|
|
423
|
+
* which pattern it matched. The rule-level `orphans.exclude` and
|
|
424
|
+
* `untested-*` `exclude` NARROW their own rule further and never re-admit a
|
|
425
|
+
* path excluded here (union, not override). Parsed once in `src/exclude.ts`.
|
|
418
426
|
*/
|
|
419
427
|
exclude?: readonly string[];
|
|
420
428
|
/**
|
package/dist/eval-cache.d.ts
CHANGED
|
@@ -6,6 +6,13 @@ export type CacheMode = "off" | "read" | "readwrite";
|
|
|
6
6
|
export interface CacheKeyInput {
|
|
7
7
|
readonly task: string;
|
|
8
8
|
readonly model: string;
|
|
9
|
+
/**
|
|
10
|
+
* Reasoning budget (`--effort`). Keyed for the same reason `model` is: it moves
|
|
11
|
+
* the output distribution, so a replay across effort levels would serve a result
|
|
12
|
+
* the caller did not ask for. `undefined` (the harness default) drops out of the
|
|
13
|
+
* hash via JSON, so entries recorded before effort existed stay valid.
|
|
14
|
+
*/
|
|
15
|
+
readonly effort?: string | number;
|
|
9
16
|
readonly tools: readonly string[];
|
|
10
17
|
/** The resolved fixture + arm + plugin files written before the run. */
|
|
11
18
|
readonly files: Record<string, string>;
|
package/dist/eval-lock.d.ts
CHANGED
|
@@ -33,6 +33,18 @@ export declare const DEFAULT_LOCK_DIR = ".vigiles/eval-locks";
|
|
|
33
33
|
export interface EvalLockInputs {
|
|
34
34
|
/** Model id used (folded in; a floating alias can't detect weight drift — warned). */
|
|
35
35
|
readonly model: string;
|
|
36
|
+
/**
|
|
37
|
+
* Reasoning budget (`--effort`) the run was pinned to, or undefined for the
|
|
38
|
+
* harness default. Hashed because it steers the model — the criterion this
|
|
39
|
+
* interface already states — so a committed report recorded at one effort is
|
|
40
|
+
* STALE for a run at another. `undefined` is dropped by `JSON.stringify`, so
|
|
41
|
+
* locks committed before effort existed keep their hash and still replay.
|
|
42
|
+
*
|
|
43
|
+
* Caveat kept honest: "omitted" means the harness's own default, which is
|
|
44
|
+
* per-model and can move between builds — reproducible only modulo that, the
|
|
45
|
+
* same class of provenance caveat as `harnessVersion` below.
|
|
46
|
+
*/
|
|
47
|
+
readonly effort?: string | number;
|
|
36
48
|
/**
|
|
37
49
|
* A hand-bumped behavior epoch the project owns (`.vigilesrc.json`
|
|
38
50
|
* `eval.apiVersion`), bumped when a harness-side change YOU made (a CLAUDE.md
|
|
@@ -71,6 +83,13 @@ export interface EvalLock {
|
|
|
71
83
|
readonly inputsHash: string;
|
|
72
84
|
/** The model id the report was produced against (for the drift warning). */
|
|
73
85
|
readonly model: string;
|
|
86
|
+
/**
|
|
87
|
+
* The effort the report was produced at, or undefined for the harness default
|
|
88
|
+
* (provenance; already in the hash). Recorded because the complaint that
|
|
89
|
+
* motivated effort support was not only that it could not be SET — it was that
|
|
90
|
+
* nothing in the run record said which effort produced the numbers.
|
|
91
|
+
*/
|
|
92
|
+
readonly effort?: string | number;
|
|
74
93
|
/** The harness version token at record time (provenance; already in the hash). */
|
|
75
94
|
readonly harnessVersionKey: string;
|
|
76
95
|
/** The behavior epoch at record time (provenance; already in the hash). */
|
|
@@ -142,6 +161,7 @@ export declare function buildLock(args: {
|
|
|
142
161
|
readonly name: string;
|
|
143
162
|
readonly inputsHash: string;
|
|
144
163
|
readonly model: string;
|
|
164
|
+
readonly effort?: string | number;
|
|
145
165
|
readonly harnessVersionKey: string;
|
|
146
166
|
readonly evalApiVersion: number;
|
|
147
167
|
readonly builtAt: string;
|
package/dist/eval.d.ts
CHANGED
|
@@ -46,6 +46,13 @@ export interface EvalArm {
|
|
|
46
46
|
* use the eval-level model. See `research/eval-architecture.md` (model strategy).
|
|
47
47
|
*/
|
|
48
48
|
readonly model?: string;
|
|
49
|
+
/**
|
|
50
|
+
* Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
|
|
51
|
+
* Lives here beside `model` because it is part of the MEASUREMENT — it moves the
|
|
52
|
+
* output distribution, not the sample size — so it is hashed into the lock and
|
|
53
|
+
* the cache, and never read from an env var. Omit for the harness default.
|
|
54
|
+
*/
|
|
55
|
+
readonly effort?: string | number;
|
|
49
56
|
}
|
|
50
57
|
/** Per-run resource use, parsed from the terminal `result` event (0 when absent). */
|
|
51
58
|
export interface EvalUsage {
|
|
@@ -95,6 +102,13 @@ export interface EvalSpec<M extends Metrics> {
|
|
|
95
102
|
readonly trials?: number;
|
|
96
103
|
/** Model alias. Default "haiku". */
|
|
97
104
|
readonly model?: string;
|
|
105
|
+
/**
|
|
106
|
+
* Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
|
|
107
|
+
* Lives here beside `model` because it is part of the MEASUREMENT — it moves the
|
|
108
|
+
* output distribution, not the sample size — so it is hashed into the lock and
|
|
109
|
+
* the cache, and never read from an env var. Omit for the harness default.
|
|
110
|
+
*/
|
|
111
|
+
readonly effort?: string | number;
|
|
98
112
|
/** Tools the agent may use. Default: Read Edit Write Bash. */
|
|
99
113
|
readonly allowedTools?: readonly string[];
|
|
100
114
|
/** Per-run timeout ms. Default 240000. */
|
|
@@ -229,6 +243,19 @@ export interface AgentRunArgs {
|
|
|
229
243
|
readonly task: string;
|
|
230
244
|
readonly cwd: string;
|
|
231
245
|
readonly model: string;
|
|
246
|
+
/**
|
|
247
|
+
* Reasoning-budget level for the run (`claude --effort`). Part of the
|
|
248
|
+
* MEASUREMENT, not a run knob: it changes the model's output distribution, not
|
|
249
|
+
* the sample size — so it lives on the spec next to `model` (never an env),
|
|
250
|
+
* and it is hashed into both the cache key and the eval lock. Deliberately
|
|
251
|
+
* `string | number` rather than a literal union: the binary accepts an alias
|
|
252
|
+
* map, is case-insensitive, and takes an integer budget, and its own valid set
|
|
253
|
+
* MOVED between builds (2.1.42 had no `xhigh`, 2.1.257 does) — a hard-coded
|
|
254
|
+
* union would reject a valid level after any upstream addition. A wrong value
|
|
255
|
+
* is caught at RUNTIME instead, by {@link effortRejection}, which is what the
|
|
256
|
+
* binary actually tells us. Omit for the harness default.
|
|
257
|
+
*/
|
|
258
|
+
readonly effort?: string | number;
|
|
232
259
|
readonly tools: readonly string[];
|
|
233
260
|
readonly hasSettings: boolean;
|
|
234
261
|
readonly pluginDir: string | undefined;
|
|
@@ -260,10 +287,74 @@ export type AgentRunner = (args: AgentRunArgs) => Promise<RunOut>;
|
|
|
260
287
|
* regression to an always-merge would otherwise silently defeat ephemerality and
|
|
261
288
|
* leak the host environment into an untrusted, model-driven run.
|
|
262
289
|
*/
|
|
263
|
-
export declare function resolveSpawnEnv(a: Pick<AgentRunArgs, "env" | "replaceEnv">, base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
264
|
-
/**
|
|
265
|
-
*
|
|
266
|
-
|
|
290
|
+
export declare function resolveSpawnEnv(a: Pick<AgentRunArgs, "env" | "replaceEnv" | "effort">, base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
291
|
+
/**
|
|
292
|
+
* The env var name the harness reads for the reasoning budget. It sits ABOVE the
|
|
293
|
+
* `--effort` flag in the CLI's own precedence chain, so passing the flag alone
|
|
294
|
+
* does NOT pin the level.
|
|
295
|
+
*/
|
|
296
|
+
export declare const EFFORT_ENV_VAR = "CLAUDE_CODE_EFFORT_LEVEL";
|
|
297
|
+
/**
|
|
298
|
+
* Pin the effort the run actually gets, so the recorded effort is the effort
|
|
299
|
+
* that ran.
|
|
300
|
+
*
|
|
301
|
+
* WHY THIS EXISTS AND WHY IT IS NOT OPTIONAL. Effort has THREE inputs — the
|
|
302
|
+
* `--effort` flag, the `effortLevel` settings key, and `CLAUDE_CODE_EFFORT_LEVEL`
|
|
303
|
+
* — and the env var wins over the flag. `EPHEMERAL_ALLOW_PREFIXES` passes
|
|
304
|
+
* `CLAUDE_*` through by design (the CLI reads several such knobs and dropping one
|
|
305
|
+
* is the failure mode), so an ambient `CLAUDE_CODE_EFFORT_LEVEL=max` in the
|
|
306
|
+
* author's shell survives even the SCRUBBED ephemeral env. Without this pin,
|
|
307
|
+
* hashing effort into the lock would make the lock CONFIDENTLY WRONG: it would
|
|
308
|
+
* record `low` over a run that executed at `max` — the exact defect the feature
|
|
309
|
+
* exists to prevent, reintroduced by the fix for it.
|
|
310
|
+
*
|
|
311
|
+
* Both directions matter, so both are handled:
|
|
312
|
+
* - effort DECLARED → set the var, overriding whatever the shell had.
|
|
313
|
+
* - effort OMITTED → DELETE an inherited var, so "omit" means the harness
|
|
314
|
+
* default rather than "whatever this machine happened to
|
|
315
|
+
* export". An omitted effort must not be a hidden input.
|
|
316
|
+
*/
|
|
317
|
+
export declare function pinEffortEnv(env: NodeJS.ProcessEnv, effort: string | number | undefined): NodeJS.ProcessEnv;
|
|
318
|
+
/**
|
|
319
|
+
* The harness's own rejection of an `--effort` value, or null. Pure.
|
|
320
|
+
*
|
|
321
|
+
* The CLI does NOT fail on a bad level — it prints this to stderr and silently
|
|
322
|
+
* runs at its default. That silent substitution is precisely the bug class this
|
|
323
|
+
* feature addresses (a number produced by a configuration nobody asked for), so
|
|
324
|
+
* a rejected value must never become a sample. Matched on the binary's own
|
|
325
|
+
* wording, the same shape as {@link isRateLimited}.
|
|
326
|
+
*/
|
|
327
|
+
export declare function effortRejection(out: RunOut): string | null;
|
|
328
|
+
/**
|
|
329
|
+
* Wrap a runner so a run the harness rejected on `--effort` FAILS LOUDLY.
|
|
330
|
+
*
|
|
331
|
+
* Applied ONCE, around the real runner, rather than as a guard repeated at each
|
|
332
|
+
* of the five `runner(...)` call sites — a guard per call site is the shape that
|
|
333
|
+
* left four of five compilers unprotected in #173.
|
|
334
|
+
*
|
|
335
|
+
* It THROWS rather than counting the trial as `runError`. A `runError` trial is
|
|
336
|
+
* dropped from the denominator, which is right for a transient (a rate limit) and
|
|
337
|
+
* wrong here: an unusable effort value is deterministic and repeatable, so every
|
|
338
|
+
* trial fails it and the run would report a rate computed over ZERO samples. A
|
|
339
|
+
* configuration mistake should stop the run and name itself.
|
|
340
|
+
*/
|
|
341
|
+
export declare function withEffortGuard(runner: AgentRunner): AgentRunner;
|
|
342
|
+
/**
|
|
343
|
+
* Build the real runner's argv. Pure and exported so the FLAGS are provable —
|
|
344
|
+
* `spawnAgentRaw` is `v8 ignore`d (it spawns a subprocess), so an argv assembled
|
|
345
|
+
* inline there could not be asserted at all. Mirrors `buildCodexArgs`.
|
|
346
|
+
*/
|
|
347
|
+
export declare function buildAgentArgs(a: AgentRunArgs): string[];
|
|
348
|
+
/**
|
|
349
|
+
* The real `claude`-spawning runner (composition root). Exported so other
|
|
350
|
+
* real-model entries (e.g. the `audit` trigger tier) bind the same runner.
|
|
351
|
+
*
|
|
352
|
+
* The effort guard is composed in HERE, at the single definition, rather than at
|
|
353
|
+
* each of the places that bind this runner — so every consumer, including ones
|
|
354
|
+
* not yet written, is covered by construction. Guarding each call site instead is
|
|
355
|
+
* the shape that left four of five compilers unprotected in #173.
|
|
356
|
+
*/
|
|
357
|
+
export declare const spawnAgent: AgentRunner;
|
|
267
358
|
/**
|
|
268
359
|
* Run the eval: every arm × every trial against the real `claude` CLI, with the
|
|
269
360
|
* metric computed per run and aggregated per arm. Requires `claude` on PATH and
|
|
@@ -320,6 +411,13 @@ export interface MeasureSpec {
|
|
|
320
411
|
readonly trials?: number;
|
|
321
412
|
/** Model alias. Default "sonnet" — measure on the model your users run. */
|
|
322
413
|
readonly model?: string;
|
|
414
|
+
/**
|
|
415
|
+
* Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
|
|
416
|
+
* Lives here beside `model` because it is part of the MEASUREMENT — it moves the
|
|
417
|
+
* output distribution, not the sample size — so it is hashed into the lock and
|
|
418
|
+
* the cache, and never read from an env var. Omit for the harness default.
|
|
419
|
+
*/
|
|
420
|
+
readonly effort?: string | number;
|
|
323
421
|
/** Tools the agent may use. */
|
|
324
422
|
readonly allowedTools?: readonly string[];
|
|
325
423
|
/** Per-run timeout ms. */
|
|
@@ -375,6 +473,13 @@ export interface ArmsMeasureSpec {
|
|
|
375
473
|
readonly model?: string;
|
|
376
474
|
readonly allowedTools?: readonly string[];
|
|
377
475
|
readonly timeoutMs?: number;
|
|
476
|
+
/**
|
|
477
|
+
* Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
|
|
478
|
+
* Lives here beside `model` because it is part of the MEASUREMENT — it moves the
|
|
479
|
+
* output distribution, not the sample size — so it is hashed into the lock and
|
|
480
|
+
* the cache, and never read from an env var. Omit for the harness default.
|
|
481
|
+
*/
|
|
482
|
+
readonly effort?: string | number;
|
|
378
483
|
readonly spacingSec?: number;
|
|
379
484
|
}
|
|
380
485
|
/** Per-arm {@link CheckReport}s — `arms[name].perCheck[i]` aligns across arms. */
|
|
@@ -471,6 +576,7 @@ export declare function runSkillSelectionTrial(args: {
|
|
|
471
576
|
readonly runner: AgentRunner;
|
|
472
577
|
readonly parse?: ModelOutputParser;
|
|
473
578
|
readonly model: string;
|
|
579
|
+
readonly effort?: string | number;
|
|
474
580
|
readonly tools?: readonly string[];
|
|
475
581
|
readonly timeoutMs?: number;
|
|
476
582
|
readonly fixture?: Record<string, string>;
|
|
@@ -658,6 +764,13 @@ export interface TriggerRateSpec {
|
|
|
658
764
|
* 0.50 on haiku vs 0.90 on Sonnet). Override for a cheaper-but-pessimistic run.
|
|
659
765
|
*/
|
|
660
766
|
readonly model?: string;
|
|
767
|
+
/**
|
|
768
|
+
* Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
|
|
769
|
+
* Lives here beside `model` because it is part of the MEASUREMENT — it moves the
|
|
770
|
+
* output distribution, not the sample size — so it is hashed into the lock and
|
|
771
|
+
* the cache, and never read from an env var. Omit for the harness default.
|
|
772
|
+
*/
|
|
773
|
+
readonly effort?: string | number;
|
|
661
774
|
/**
|
|
662
775
|
* Minimum model tier this eval may run on (haiku<sonnet<opus by family). The
|
|
663
776
|
* run **fails** if the resolved `model` is weaker — trigger-rate under-measures
|