vigiles 25.0.0 → 26.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -36,7 +36,7 @@ function readNpmScripts(basePath) {
36
36
  * Collect commands documented in specs by loading spec source files directly.
37
37
  * Reads the structured `commands` field — no markdown parsing.
38
38
  */
39
- function collectDocumentedCommands(basePath, specs) {
39
+ function collectDocumentedCommands(basePath, specs, ignore) {
40
40
  const commands = new Set();
41
41
  if (specs) {
42
42
  for (const spec of specs) {
@@ -50,8 +50,12 @@ function collectDocumentedCommands(basePath, specs) {
50
50
  // Fallback: scan compiled markdown for spec file references, then
51
51
  // load the compiled JS spec from dist/. If that fails, try to
52
52
  // extract commands from the compiled output (last resort).
53
+ // `ignore` is the repo's ExcludeSet string face (src/exclude.ts): the floor
54
+ // plus `.vigilesrc.json#exclude`, required so this fallback cannot walk a
55
+ // vendored corpus the repo excluded (#192). The CLI always passes `specs`,
56
+ // so this branch is reached by the library path and tests only.
53
57
  const mdFiles = (0, glob_1.globSync)("**/*.md", {
54
- ignore: ["node_modules/**", "dist/**", ".vigiles/**"],
58
+ ignore: [...ignore],
55
59
  cwd: basePath,
56
60
  });
57
61
  for (const mdFile of mdFiles) {
@@ -92,9 +96,9 @@ function collectDocumentedCommands(basePath, specs) {
92
96
  /**
93
97
  * Compute script coverage: what % of npm scripts are documented in specs.
94
98
  */
95
- function computeScriptCoverage(basePath, threshold, specs) {
99
+ function computeScriptCoverage(basePath, threshold, specs, ignore) {
96
100
  const allScripts = readNpmScripts(basePath);
97
- const documented = collectDocumentedCommands(basePath, specs);
101
+ const documented = collectDocumentedCommands(basePath, specs, ignore);
98
102
  const covered = [];
99
103
  const uncovered = [];
100
104
  for (const script of allScripts) {
@@ -146,13 +150,13 @@ function computeLinterRuleCoverage(enabled, documented, threshold) {
146
150
  /**
147
151
  * Check all coverage metrics against thresholds.
148
152
  */
149
- function checkCoverage(basePath, thresholds, linterEnabled, linterDocumented, specs) {
153
+ function checkCoverage(basePath, thresholds, linterEnabled, linterDocumented, specs, ignore) {
150
154
  const metrics = [];
151
155
  // Linter rule coverage
152
156
  const linterMetric = computeLinterRuleCoverage(linterEnabled, linterDocumented, thresholds.linterRules);
153
157
  metrics.push(linterMetric);
154
158
  // Script coverage
155
- const scriptMetric = computeScriptCoverage(basePath, thresholds.scripts, specs);
159
+ const scriptMetric = computeScriptCoverage(basePath, thresholds.scripts, specs, ignore);
156
160
  metrics.push(scriptMetric);
157
161
  const passing = metrics.every((m) => m.passing);
158
162
  return { metrics, passing };
@@ -489,7 +489,48 @@ function commandView(raw, root) {
489
489
  raw,
490
490
  runs(program, opts) {
491
491
  const tokens = program.split(/\s+/).filter(Boolean);
492
- return leaves.some((argv) => runsSeq(argv, tokens) && (opts?.force ? hasForce(argv) : true));
492
+ // 🔴 READS RAW **AND** NORMALIZED ARGV — the same union `touches` already
493
+ // uses, and for the same reason. This matcher read only `leaves` (raw), so
494
+ // three ordinary rewrites walked past a guard that blocks the plain form.
495
+ // MEASURED 2026-09-02 on the shipped dogfood artifact behind the public
496
+ // 7/7 claim (`examples/harness/safe-bash-guard.mjs`):
497
+ // git push --force origin main exit 2 blocked
498
+ // git push "--force" origin main exit 0 ALLOWED
499
+ // sudo git push --force origin main exit 0 ALLOWED
500
+ // /usr/bin/git push --force origin main exit 0 ALLOWED
501
+ // Across 30 shell-equivalent variants of the seven catalog seeds the guard
502
+ // blocked 8. The seeds were 7/7 — the number was true and measured on the
503
+ // only forms anyone had written down.
504
+ //
505
+ // The quoting case is the sharpest: `getLiteral` (bash-effects) returns
506
+ // null for a quoted word, and `leafCommands` FILTERS nulls, so
507
+ // `git push "--force" origin main` arrives as [git, push, origin, main] —
508
+ // a NON-force push. The force check was answering correctly about an argv
509
+ // that had already lost the flag.
510
+ //
511
+ // Force is asked of the NORMALIZED leaf via `hasFlag`, not by re-scanning
512
+ // strings: short clusters are already split (`-rf`→r,f) and each flag is
513
+ // recorded in both short and long form, so `-f` and `--force` are one
514
+ // question. The raw half stays because it is what an exact-literal match
515
+ // is for; neither half alone is enough, which is exactly what the comment
516
+ // on `touches` says about its own union.
517
+ const rawHit = leaves.some((argv) => runsSeq(argv, tokens) && (opts?.force ? hasForce(argv) : true));
518
+ if (rawHit)
519
+ return true;
520
+ // A FLAG named in the pattern is asked of `hasFlag`, not matched as a
521
+ // literal token. The normalized leaf already records every flag in BOTH
522
+ // short and long form, so `runs("git commit --no-verify")` also catches
523
+ // `git commit -n` — which the literal path cannot, because `-n` is simply
524
+ // a different string. Found by the generated battery on its first real
525
+ // run: 72/73, and the one miss was exactly this
526
+ // (`git commit -n -m 'skip hooks'`). The `{force:true}` option exists
527
+ // because force needed this for `-f`; every other flag needed it too and
528
+ // nobody noticed, since nobody wrote the short form down by hand.
529
+ const flagTokens = tokens.filter((t) => t.startsWith("-"));
530
+ const plainTokens = tokens.filter((t) => !t.startsWith("-"));
531
+ return normalized.some((leaf) => runsSeq(leaf.argv, plainTokens) &&
532
+ flagTokens.every((t) => leaf.hasFlag(t.replace(/^-+/, ""))) &&
533
+ (opts?.force ? leaf.hasFlag("force", "f") : true));
493
534
  },
494
535
  isSideEffecting: () => (0, bash_effects_js_1.classifyBashCommand)(raw) === "side-effecting",
495
536
  // Both of these are DENYLIST matchers, so both break an undecidable verdict
@@ -532,7 +573,12 @@ function commandView(raw, root) {
532
573
  // normalized ones, and a gate built on both had a silent hole).
533
574
  writesTo: (prefixes) => matchedWriteTargets(prefixes).length > 0,
534
575
  writeTargets: matchedWriteTargets,
535
- pipesToShell: () => leaves.some(isBareShellLeaf),
576
+ // Same union, same reason: `curl … | /bin/sh` and `curl … | sudo sh` are a
577
+ // pipe-to-shell, and a raw-only check misses both (`isBareShellLeaf` tests
578
+ // the head literally, and the normalizer is what reduces `/bin/sh`→`sh` and
579
+ // strips the wrapper).
580
+ pipesToShell: () => leaves.some(isBareShellLeaf) ||
581
+ normalized.some((leaf) => isBareShellLeaf(leaf.argv)),
536
582
  };
537
583
  }
538
584
  const tool = (name) => ({ tool: name });
@@ -32,8 +32,15 @@ export interface FindOrphansOptions {
32
32
  * or to your project's doc globs (e.g. `["wiki/**\/*.md"]`) to override.
33
33
  */
34
34
  readonly include?: readonly string[];
35
- /** Glob patterns to exclude within the include scope. */
35
+ /** Glob patterns to exclude within the include scope (orphan CANDIDACY only). */
36
36
  readonly exclude?: readonly string[];
37
+ /**
38
+ * The repo-wide `.vigilesrc.json#exclude` (the ExcludeSet string face,
39
+ * src/exclude.ts). Applied to BOTH walks — candidates AND the reference scan —
40
+ * so an excluded corpus can neither be an orphan nor keep one alive (#192).
41
+ * The CLI always passes it; a direct library caller may omit it.
42
+ */
43
+ readonly repoExclude?: readonly string[];
37
44
  /**
38
45
  * Harnesses whose surface files (instruction file, `SKILL.md`, subagents,
39
46
  * commands) are load-bearing by location and thus never orphan CANDIDATES
@@ -153,12 +153,18 @@ function findOrphanDocs(options = {}) {
153
153
  const basePath = options.basePath ?? process.cwd();
154
154
  const include = options.include ?? DEFAULT_INCLUDE;
155
155
  const userExclude = options.exclude ?? [];
156
- const ignore = [...DEFAULT_IGNORE, ...userExclude];
156
+ // Two exclusions with two jobs (#192). `repoExclude` is the repo-wide
157
+ // `.vigilesrc.json#exclude` — a vendored corpus is neither an orphan
158
+ // CANDIDATE nor a SOURCE of references (a link from inside it must not keep a
159
+ // doc alive). The rule's own `exclude` only narrows candidacy: a doc kept out
160
+ // of the orphan list can still reference others. Union, never override.
161
+ const repoExclude = options.repoExclude ?? [];
162
+ const ignore = [...DEFAULT_IGNORE, ...repoExclude, ...userExclude];
157
163
  const layouts = options.layouts ?? [];
158
164
  const allDocs = collectDocs(basePath, include, ignore, layouts);
159
165
  const allMarkdown = (0, glob_1.globSync)("**/*.md", {
160
166
  cwd: basePath,
161
- ignore: [...DEFAULT_IGNORE],
167
+ ignore: [...DEFAULT_IGNORE, ...repoExclude],
162
168
  });
163
169
  const referencedBy = new Map();
164
170
  for (const mdPath of allMarkdown) {
@@ -409,12 +409,20 @@ export interface VigilesConfig {
409
409
  bundles?: "root" | "all";
410
410
  orphans?: OrphansConfig;
411
411
  /**
412
- * Glob patterns of instruction/skill files to EXCLUDE from `lint` discovery
413
- * (tsconfig-style, relative to the repo root). Use it for vendored or
414
- * benchmark fixtures the repo's own lint shouldn't police — e.g.
415
- * `["bench/**"]` so a third-party `CLAUDE.md` injected verbatim as a benchmark
416
- * arm isn't held to `require-instructions-spec`. `node_modules`/`dist` are always
417
- * excluded.
412
+ * Paths and globs the repo's own tooling does NOT police — vendored corpora,
413
+ * benchmark fixtures, frozen reproductions (tsconfig-style, relative to the
414
+ * repo root; a bare directory name such as `"bench"` excludes its subtree, as
415
+ * do `"bench/"` and `"bench/**"`). `node_modules`/`dist`/`.git`/`.vigiles` are
416
+ * always excluded.
417
+ *
418
+ * ONE filter, every pass (#192): `compile` does not load an excluded spec,
419
+ * `lint` does not discover an excluded instruction file, nested bundle, doc, or
420
+ * surface, `audit` does not read an excluded instruction file, and
421
+ * `test`/`eval` do not discover an excluded script. It filters DISCOVERY only:
422
+ * a path you name on the command line is still processed, and one line says
423
+ * which pattern it matched. The rule-level `orphans.exclude` and
424
+ * `untested-*` `exclude` NARROW their own rule further and never re-admit a
425
+ * path excluded here (union, not override). Parsed once in `src/exclude.ts`.
418
426
  */
419
427
  exclude?: readonly string[];
420
428
  /**
@@ -6,6 +6,13 @@ export type CacheMode = "off" | "read" | "readwrite";
6
6
  export interface CacheKeyInput {
7
7
  readonly task: string;
8
8
  readonly model: string;
9
+ /**
10
+ * Reasoning budget (`--effort`). Keyed for the same reason `model` is: it moves
11
+ * the output distribution, so a replay across effort levels would serve a result
12
+ * the caller did not ask for. `undefined` (the harness default) drops out of the
13
+ * hash via JSON, so entries recorded before effort existed stay valid.
14
+ */
15
+ readonly effort?: string | number;
9
16
  readonly tools: readonly string[];
10
17
  /** The resolved fixture + arm + plugin files written before the run. */
11
18
  readonly files: Record<string, string>;
@@ -33,6 +33,18 @@ export declare const DEFAULT_LOCK_DIR = ".vigiles/eval-locks";
33
33
  export interface EvalLockInputs {
34
34
  /** Model id used (folded in; a floating alias can't detect weight drift — warned). */
35
35
  readonly model: string;
36
+ /**
37
+ * Reasoning budget (`--effort`) the run was pinned to, or undefined for the
38
+ * harness default. Hashed because it steers the model — the criterion this
39
+ * interface already states — so a committed report recorded at one effort is
40
+ * STALE for a run at another. `undefined` is dropped by `JSON.stringify`, so
41
+ * locks committed before effort existed keep their hash and still replay.
42
+ *
43
+ * Caveat kept honest: "omitted" means the harness's own default, which is
44
+ * per-model and can move between builds — reproducible only modulo that, the
45
+ * same class of provenance caveat as `harnessVersion` below.
46
+ */
47
+ readonly effort?: string | number;
36
48
  /**
37
49
  * A hand-bumped behavior epoch the project owns (`.vigilesrc.json`
38
50
  * `eval.apiVersion`), bumped when a harness-side change YOU made (a CLAUDE.md
@@ -71,6 +83,13 @@ export interface EvalLock {
71
83
  readonly inputsHash: string;
72
84
  /** The model id the report was produced against (for the drift warning). */
73
85
  readonly model: string;
86
+ /**
87
+ * The effort the report was produced at, or undefined for the harness default
88
+ * (provenance; already in the hash). Recorded because the complaint that
89
+ * motivated effort support was not only that it could not be SET — it was that
90
+ * nothing in the run record said which effort produced the numbers.
91
+ */
92
+ readonly effort?: string | number;
74
93
  /** The harness version token at record time (provenance; already in the hash). */
75
94
  readonly harnessVersionKey: string;
76
95
  /** The behavior epoch at record time (provenance; already in the hash). */
@@ -142,6 +161,7 @@ export declare function buildLock(args: {
142
161
  readonly name: string;
143
162
  readonly inputsHash: string;
144
163
  readonly model: string;
164
+ readonly effort?: string | number;
145
165
  readonly harnessVersionKey: string;
146
166
  readonly evalApiVersion: number;
147
167
  readonly builtAt: string;
package/dist/eval.d.ts CHANGED
@@ -46,6 +46,13 @@ export interface EvalArm {
46
46
  * use the eval-level model. See `research/eval-architecture.md` (model strategy).
47
47
  */
48
48
  readonly model?: string;
49
+ /**
50
+ * Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
51
+ * Lives here beside `model` because it is part of the MEASUREMENT — it moves the
52
+ * output distribution, not the sample size — so it is hashed into the lock and
53
+ * the cache, and never read from an env var. Omit for the harness default.
54
+ */
55
+ readonly effort?: string | number;
49
56
  }
50
57
  /** Per-run resource use, parsed from the terminal `result` event (0 when absent). */
51
58
  export interface EvalUsage {
@@ -95,6 +102,13 @@ export interface EvalSpec<M extends Metrics> {
95
102
  readonly trials?: number;
96
103
  /** Model alias. Default "haiku". */
97
104
  readonly model?: string;
105
+ /**
106
+ * Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
107
+ * Lives here beside `model` because it is part of the MEASUREMENT — it moves the
108
+ * output distribution, not the sample size — so it is hashed into the lock and
109
+ * the cache, and never read from an env var. Omit for the harness default.
110
+ */
111
+ readonly effort?: string | number;
98
112
  /** Tools the agent may use. Default: Read Edit Write Bash. */
99
113
  readonly allowedTools?: readonly string[];
100
114
  /** Per-run timeout ms. Default 240000. */
@@ -229,6 +243,19 @@ export interface AgentRunArgs {
229
243
  readonly task: string;
230
244
  readonly cwd: string;
231
245
  readonly model: string;
246
+ /**
247
+ * Reasoning-budget level for the run (`claude --effort`). Part of the
248
+ * MEASUREMENT, not a run knob: it changes the model's output distribution, not
249
+ * the sample size — so it lives on the spec next to `model` (never an env),
250
+ * and it is hashed into both the cache key and the eval lock. Deliberately
251
+ * `string | number` rather than a literal union: the binary accepts an alias
252
+ * map, is case-insensitive, and takes an integer budget, and its own valid set
253
+ * MOVED between builds (2.1.42 had no `xhigh`, 2.1.257 does) — a hard-coded
254
+ * union would reject a valid level after any upstream addition. A wrong value
255
+ * is caught at RUNTIME instead, by {@link effortRejection}, which is what the
256
+ * binary actually tells us. Omit for the harness default.
257
+ */
258
+ readonly effort?: string | number;
232
259
  readonly tools: readonly string[];
233
260
  readonly hasSettings: boolean;
234
261
  readonly pluginDir: string | undefined;
@@ -260,10 +287,74 @@ export type AgentRunner = (args: AgentRunArgs) => Promise<RunOut>;
260
287
  * regression to an always-merge would otherwise silently defeat ephemerality and
261
288
  * leak the host environment into an untrusted, model-driven run.
262
289
  */
263
- export declare function resolveSpawnEnv(a: Pick<AgentRunArgs, "env" | "replaceEnv">, base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
264
- /** The real `claude`-spawning runner (composition root). Exported so other
265
- * real-model entries (e.g. the `audit` trigger tier) bind the same runner. */
266
- export declare function spawnAgent(a: AgentRunArgs): Promise<RunOut>;
290
+ export declare function resolveSpawnEnv(a: Pick<AgentRunArgs, "env" | "replaceEnv" | "effort">, base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
291
+ /**
292
+ * The env var name the harness reads for the reasoning budget. It sits ABOVE the
293
+ * `--effort` flag in the CLI's own precedence chain, so passing the flag alone
294
+ * does NOT pin the level.
295
+ */
296
+ export declare const EFFORT_ENV_VAR = "CLAUDE_CODE_EFFORT_LEVEL";
297
+ /**
298
+ * Pin the effort the run actually gets, so the recorded effort is the effort
299
+ * that ran.
300
+ *
301
+ * WHY THIS EXISTS AND WHY IT IS NOT OPTIONAL. Effort has THREE inputs — the
302
+ * `--effort` flag, the `effortLevel` settings key, and `CLAUDE_CODE_EFFORT_LEVEL`
303
+ * — and the env var wins over the flag. `EPHEMERAL_ALLOW_PREFIXES` passes
304
+ * `CLAUDE_*` through by design (the CLI reads several such knobs and dropping one
305
+ * is the failure mode), so an ambient `CLAUDE_CODE_EFFORT_LEVEL=max` in the
306
+ * author's shell survives even the SCRUBBED ephemeral env. Without this pin,
307
+ * hashing effort into the lock would make the lock CONFIDENTLY WRONG: it would
308
+ * record `low` over a run that executed at `max` — the exact defect the feature
309
+ * exists to prevent, reintroduced by the fix for it.
310
+ *
311
+ * Both directions matter, so both are handled:
312
+ * - effort DECLARED → set the var, overriding whatever the shell had.
313
+ * - effort OMITTED → DELETE an inherited var, so "omit" means the harness
314
+ * default rather than "whatever this machine happened to
315
+ * export". An omitted effort must not be a hidden input.
316
+ */
317
+ export declare function pinEffortEnv(env: NodeJS.ProcessEnv, effort: string | number | undefined): NodeJS.ProcessEnv;
318
+ /**
319
+ * The harness's own rejection of an `--effort` value, or null. Pure.
320
+ *
321
+ * The CLI does NOT fail on a bad level — it prints this to stderr and silently
322
+ * runs at its default. That silent substitution is precisely the bug class this
323
+ * feature addresses (a number produced by a configuration nobody asked for), so
324
+ * a rejected value must never become a sample. Matched on the binary's own
325
+ * wording, the same shape as {@link isRateLimited}.
326
+ */
327
+ export declare function effortRejection(out: RunOut): string | null;
328
+ /**
329
+ * Wrap a runner so a run the harness rejected on `--effort` FAILS LOUDLY.
330
+ *
331
+ * Applied ONCE, around the real runner, rather than as a guard repeated at each
332
+ * of the five `runner(...)` call sites — a guard per call site is the shape that
333
+ * left four of five compilers unprotected in #173.
334
+ *
335
+ * It THROWS rather than counting the trial as `runError`. A `runError` trial is
336
+ * dropped from the denominator, which is right for a transient (a rate limit) and
337
+ * wrong here: an unusable effort value is deterministic and repeatable, so every
338
+ * trial fails it and the run would report a rate computed over ZERO samples. A
339
+ * configuration mistake should stop the run and name itself.
340
+ */
341
+ export declare function withEffortGuard(runner: AgentRunner): AgentRunner;
342
+ /**
343
+ * Build the real runner's argv. Pure and exported so the FLAGS are provable —
344
+ * `spawnAgentRaw` is `v8 ignore`d (it spawns a subprocess), so an argv assembled
345
+ * inline there could not be asserted at all. Mirrors `buildCodexArgs`.
346
+ */
347
+ export declare function buildAgentArgs(a: AgentRunArgs): string[];
348
+ /**
349
+ * The real `claude`-spawning runner (composition root). Exported so other
350
+ * real-model entries (e.g. the `audit` trigger tier) bind the same runner.
351
+ *
352
+ * The effort guard is composed in HERE, at the single definition, rather than at
353
+ * each of the places that bind this runner — so every consumer, including ones
354
+ * not yet written, is covered by construction. Guarding each call site instead is
355
+ * the shape that left four of five compilers unprotected in #173.
356
+ */
357
+ export declare const spawnAgent: AgentRunner;
267
358
  /**
268
359
  * Run the eval: every arm × every trial against the real `claude` CLI, with the
269
360
  * metric computed per run and aggregated per arm. Requires `claude` on PATH and
@@ -320,6 +411,13 @@ export interface MeasureSpec {
320
411
  readonly trials?: number;
321
412
  /** Model alias. Default "sonnet" — measure on the model your users run. */
322
413
  readonly model?: string;
414
+ /**
415
+ * Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
416
+ * Lives here beside `model` because it is part of the MEASUREMENT — it moves the
417
+ * output distribution, not the sample size — so it is hashed into the lock and
418
+ * the cache, and never read from an env var. Omit for the harness default.
419
+ */
420
+ readonly effort?: string | number;
323
421
  /** Tools the agent may use. */
324
422
  readonly allowedTools?: readonly string[];
325
423
  /** Per-run timeout ms. */
@@ -375,6 +473,13 @@ export interface ArmsMeasureSpec {
375
473
  readonly model?: string;
376
474
  readonly allowedTools?: readonly string[];
377
475
  readonly timeoutMs?: number;
476
+ /**
477
+ * Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
478
+ * Lives here beside `model` because it is part of the MEASUREMENT — it moves the
479
+ * output distribution, not the sample size — so it is hashed into the lock and
480
+ * the cache, and never read from an env var. Omit for the harness default.
481
+ */
482
+ readonly effort?: string | number;
378
483
  readonly spacingSec?: number;
379
484
  }
380
485
  /** Per-arm {@link CheckReport}s — `arms[name].perCheck[i]` aligns across arms. */
@@ -471,6 +576,7 @@ export declare function runSkillSelectionTrial(args: {
471
576
  readonly runner: AgentRunner;
472
577
  readonly parse?: ModelOutputParser;
473
578
  readonly model: string;
579
+ readonly effort?: string | number;
474
580
  readonly tools?: readonly string[];
475
581
  readonly timeoutMs?: number;
476
582
  readonly fixture?: Record<string, string>;
@@ -658,6 +764,13 @@ export interface TriggerRateSpec {
658
764
  * 0.50 on haiku vs 0.90 on Sonnet). Override for a cheaper-but-pessimistic run.
659
765
  */
660
766
  readonly model?: string;
767
+ /**
768
+ * Reasoning budget for the run (`claude --effort`, e.g. `"low"` or an integer).
769
+ * Lives here beside `model` because it is part of the MEASUREMENT — it moves the
770
+ * output distribution, not the sample size — so it is hashed into the lock and
771
+ * the cache, and never read from an env var. Omit for the harness default.
772
+ */
773
+ readonly effort?: string | number;
661
774
  /**
662
775
  * Minimum model tier this eval may run on (haiku<sonnet<opus by family). The
663
776
  * run **fails** if the resolved `model` is weaker — trigger-rate under-measures