vigiles 15.0.2 → 15.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/adapters/claude-code/layout.js +3 -0
  2. package/dist/adapters/claude-code/run-scripts.d.ts +21 -9
  3. package/dist/adapters/claude-code/run-scripts.js +33 -22
  4. package/dist/adapters/codex/driver.js +3 -1
  5. package/dist/adapters/codex/eval.js +5 -0
  6. package/dist/audit-score.js +32 -6
  7. package/dist/check-count.d.ts +62 -3
  8. package/dist/check-count.js +123 -5
  9. package/dist/check.d.ts +55 -0
  10. package/dist/check.js +91 -0
  11. package/dist/cli.js +456 -54
  12. package/dist/core/bash-effects.d.ts +41 -0
  13. package/dist/core/bash-effects.js +278 -0
  14. package/dist/core/foreign-runner.d.ts +185 -0
  15. package/dist/core/foreign-runner.js +228 -0
  16. package/dist/core/harness-driver.d.ts +18 -1
  17. package/dist/core/hook-program.d.ts +319 -20
  18. package/dist/core/hook-program.js +743 -49
  19. package/dist/core/layout.d.ts +13 -0
  20. package/dist/core/lethal-trifecta.d.ts +93 -0
  21. package/dist/core/lethal-trifecta.js +409 -65
  22. package/dist/core/markdown.d.ts +32 -0
  23. package/dist/core/markdown.js +36 -0
  24. package/dist/core/merge-conflict.d.ts +50 -0
  25. package/dist/core/merge-conflict.js +84 -0
  26. package/dist/core/test-file-ext.d.ts +80 -0
  27. package/dist/core/test-file-ext.js +90 -0
  28. package/dist/core/types.d.ts +11 -0
  29. package/dist/coverage-artifact.d.ts +383 -0
  30. package/dist/coverage-artifact.js +586 -0
  31. package/dist/coverage-evidence.d.ts +87 -79
  32. package/dist/coverage-evidence.js +212 -211
  33. package/dist/coverage-probe.d.ts +125 -0
  34. package/dist/coverage-probe.js +700 -0
  35. package/dist/doc-commands.d.ts +119 -0
  36. package/dist/doc-commands.js +158 -0
  37. package/dist/eval-lock.d.ts +9 -0
  38. package/dist/eval-lock.js +14 -2
  39. package/dist/eval.js +28 -0
  40. package/dist/fs-walk.d.ts +80 -0
  41. package/dist/fs-walk.js +148 -0
  42. package/dist/harness-assert.js +42 -3
  43. package/dist/harness-test.js +31 -2
  44. package/dist/judge.js +4 -0
  45. package/dist/leaderboard.d.ts +16 -1
  46. package/dist/leaderboard.js +19 -2
  47. package/dist/load-hook.js +6 -1
  48. package/dist/mock-model.d.ts +58 -4
  49. package/dist/mock-model.js +133 -5
  50. package/dist/observe.d.ts +1 -1
  51. package/dist/observe.js +44 -1
  52. package/dist/plugin-loader.js +52 -13
  53. package/dist/run-hook.d.ts +58 -0
  54. package/dist/run-hook.js +70 -0
  55. package/dist/run-script.js +75 -0
  56. package/dist/scaffold-test.d.ts +3 -0
  57. package/dist/scaffold-test.js +10 -5
  58. package/dist/scan-behavioral.js +4 -0
  59. package/dist/scan-core.d.ts +22 -0
  60. package/dist/scan-core.js +46 -14
  61. package/dist/scan-files.js +19 -2
  62. package/dist/scan.d.ts +22 -5
  63. package/dist/scan.js +67 -13
  64. package/dist/skill-contract.d.ts +46 -0
  65. package/dist/skill-contract.js +201 -0
  66. package/dist/skill-refs.d.ts +67 -0
  67. package/dist/skill-refs.js +117 -0
  68. package/dist/test-coverage-files.d.ts +7 -1
  69. package/dist/test-coverage-files.js +66 -55
  70. package/dist/test-coverage.d.ts +155 -19
  71. package/dist/test-coverage.js +430 -91
  72. package/dist/testing.d.ts +8 -2
  73. package/dist/testing.js +19 -1
  74. package/dist/trigger-containment.d.ts +85 -0
  75. package/dist/trigger-containment.js +125 -0
  76. package/dist/ts-runner-caps.d.ts +16 -0
  77. package/dist/ts-runner-caps.js +43 -0
  78. package/dist/unit.d.ts +2 -2
  79. package/dist/unit.js +2 -1
  80. package/hooks/eval-lock-nudge.sh +14 -6
  81. package/package.json +2 -2
  82. package/skills/test-harness/SKILL.md +68 -2
@@ -131,4 +131,45 @@ export interface NormalizedLeaf {
131
131
  * failure → []. A leaf with a dynamic head is skipped (can't be normalized).
132
132
  */
133
133
  export declare function leafCommandsNormalized(command: string): NormalizedLeaf[];
134
+ /**
135
+ * Every simple command's argv, POSITIONALLY, each word reconstructed at source
136
+ * level — the primitive a caller needs to tell an EXECUTED PROGRAM from a DATA
137
+ * OPERAND.
138
+ *
139
+ * 🔴 WHY POSITION, AND WHY A THIRD EXTRACTOR. `leafCommands` DROPS the words it
140
+ * cannot reduce to a literal, so `"$GUARD" --flag` yields `["--flag"]` — a
141
+ * dynamic head silently promotes an argument into head position, which is worse
142
+ * than useless to a positional reader. `leafCommandsNormalized` skips such a leaf
143
+ * outright AND basenames the head, so `./hooks/x.sh` loses the path a file
144
+ * resolver needs. This one keeps every word in its slot (an unreconstructable
145
+ * word becomes `""`) and keeps the head's spelling.
146
+ *
147
+ * Wrappers are still resolved through (`env FOO=1 bash x.sh` → `bash x.sh`)
148
+ * using the same wrapper table as the normalized extractor, not a second copy.
149
+ *
150
+ * 🔴 ONLY THE LEAVES THAT UNCONDITIONALLY RUN, and this used to be a blanket
151
+ * `Walk` over every `CallExpr` in the tree. A syntactic leaf is not an executed
152
+ * command: `false && bash hooks/pre.sh` and `true || bash hooks/pre.sh` never run
153
+ * the hook, and `if false; then bash hooks/x.sh; fi` and a function BODY do not
154
+ * run at all where they are written — yet all four were reported as executed
155
+ * programs. For the one caller (coverage attribution) that is a FALSE GRANT: a
156
+ * hook credited with a run that never happened. Same class as attributing a data
157
+ * operand, one level up in the grammar.
158
+ *
159
+ * So the tree is DESCENDED, not walked, and only through positions whose children
160
+ * always execute: a statement list, both sides of a PIPELINE, a subshell, a
161
+ * block. `&&`/`||` contribute their LEFT side only — the right is conditional.
162
+ * Every other construct (`if`, `while`, `until`, `for`, `case`, a function
163
+ * declaration, `time`, …) is not entered at all: whether its body ran is a
164
+ * runtime fact this parse cannot have, and abstaining costs one warning while
165
+ * guessing costs a false claim that something was tested.
166
+ *
167
+ * ⚠️ THE MEASURED COST, stated rather than assumed: `cd /repo && bash hooks/x.sh`
168
+ * now yields NOTHING, because the hook sits on the right of `&&`. That idiom has
169
+ * a first-class replacement — `runHook(cmd, event, { cwd })` — which is how this
170
+ * repo's own examples already write it.
171
+ *
172
+ * Parse failure → `[]`.
173
+ */
174
+ export declare function leafArgvSource(command: string): string[][];
134
175
  //# sourceMappingURL=bash-effects.d.ts.map
@@ -28,6 +28,7 @@ exports.classifyBashCommand = classifyBashCommand;
28
28
  exports.isReadOnlyBash = isReadOnlyBash;
29
29
  exports.leafCommands = leafCommands;
30
30
  exports.leafCommandsNormalized = leafCommandsNormalized;
31
+ exports.leafArgvSource = leafArgvSource;
31
32
  // mvdan-sh is a CJS package (GopherJS build) with no bundled TypeScript types.
32
33
  // The project compiles to CommonJS (Node16, no "type":"module"), so plain
33
34
  // require() works and is the idiomatic pattern here (see linters.ts).
@@ -719,6 +720,283 @@ function leafCommandsNormalized(command) {
719
720
  });
720
721
  return out;
721
722
  }
723
+ /**
724
+ * Reconstruct a Word's SOURCE-level text: quotes unwrapped, parameter references
725
+ * kept verbatim (`${CLAUDE_PROJECT_DIR}` → `$CLAUDE_PROJECT_DIR`). `null` when a
726
+ * segment cannot be reconstructed at all (command substitution, arithmetic,
727
+ * process substitution).
728
+ *
729
+ * The twin of {@link normalizeParts}, and deliberately NOT the same function.
730
+ * `normalizeParts` answers "what OPERATION is this" and therefore collapses
731
+ * `$HOME` to `~` and gives up on every other parameter; a caller that needs the
732
+ * PATH a word names can use neither behaviour — `$CLAUDE_PROJECT_DIR/.claude/hooks/x.sh`
733
+ * has to survive as written, because a file resolver matches it by suffix.
734
+ */
735
+ function sourceParts(parts) {
736
+ if (!parts)
737
+ return null;
738
+ let out = "";
739
+ for (const p of parts) {
740
+ const t = sh.syntax.NodeType(p);
741
+ if (t === "Lit" || t === "SglQuoted") {
742
+ out += p.Value ?? "";
743
+ }
744
+ else if (t === "DblQuoted") {
745
+ const inner = sourceParts(p.Parts);
746
+ if (inner === null)
747
+ return null;
748
+ out += inner;
749
+ }
750
+ else if (t === "ParamExp") {
751
+ const name = p.Param?.Value;
752
+ if (!name)
753
+ return null;
754
+ out += `$${name}`;
755
+ }
756
+ else {
757
+ return null; // CmdSubst / ArithmExp / ProcSubst / … → unreconstructable
758
+ }
759
+ }
760
+ return out;
761
+ }
762
+ /**
763
+ * Every simple command's argv, POSITIONALLY, each word reconstructed at source
764
+ * level — the primitive a caller needs to tell an EXECUTED PROGRAM from a DATA
765
+ * OPERAND.
766
+ *
767
+ * 🔴 WHY POSITION, AND WHY A THIRD EXTRACTOR. `leafCommands` DROPS the words it
768
+ * cannot reduce to a literal, so `"$GUARD" --flag` yields `["--flag"]` — a
769
+ * dynamic head silently promotes an argument into head position, which is worse
770
+ * than useless to a positional reader. `leafCommandsNormalized` skips such a leaf
771
+ * outright AND basenames the head, so `./hooks/x.sh` loses the path a file
772
+ * resolver needs. This one keeps every word in its slot (an unreconstructable
773
+ * word becomes `""`) and keeps the head's spelling.
774
+ *
775
+ * Wrappers are still resolved through (`env FOO=1 bash x.sh` → `bash x.sh`)
776
+ * using the same wrapper table as the normalized extractor, not a second copy.
777
+ *
778
+ * 🔴 ONLY THE LEAVES THAT UNCONDITIONALLY RUN, and this used to be a blanket
779
+ * `Walk` over every `CallExpr` in the tree. A syntactic leaf is not an executed
780
+ * command: `false && bash hooks/pre.sh` and `true || bash hooks/pre.sh` never run
781
+ * the hook, and `if false; then bash hooks/x.sh; fi` and a function BODY do not
782
+ * run at all where they are written — yet all four were reported as executed
783
+ * programs. For the one caller (coverage attribution) that is a FALSE GRANT: a
784
+ * hook credited with a run that never happened. Same class as attributing a data
785
+ * operand, one level up in the grammar.
786
+ *
787
+ * So the tree is DESCENDED, not walked, and only through positions whose children
788
+ * always execute: a statement list, both sides of a PIPELINE, a subshell, a
789
+ * block. `&&`/`||` contribute their LEFT side only — the right is conditional.
790
+ * Every other construct (`if`, `while`, `until`, `for`, `case`, a function
791
+ * declaration, `time`, …) is not entered at all: whether its body ran is a
792
+ * runtime fact this parse cannot have, and abstaining costs one warning while
793
+ * guessing costs a false claim that something was tested.
794
+ *
795
+ * ⚠️ THE MEASURED COST, stated rather than assumed: `cd /repo && bash hooks/x.sh`
796
+ * now yields NOTHING, because the hook sits on the right of `&&`. That idiom has
797
+ * a first-class replacement — `runHook(cmd, event, { cwd })` — which is how this
798
+ * repo's own examples already write it.
799
+ *
800
+ * Parse failure → `[]`.
801
+ */
802
+ function leafArgvSource(command) {
803
+ let file;
804
+ try {
805
+ file = sh.syntax.NewParser().Parse(command, "cmd.sh");
806
+ }
807
+ catch {
808
+ return [];
809
+ }
810
+ const out = [];
811
+ const emit = (node) => {
812
+ if (!node.Args?.length)
813
+ return;
814
+ const argv = node.Args.map((w) => sourceParts(w.Parts) ?? "");
815
+ // `command -v x` DESCRIBES x; it does not run it. Without this the wrapper
816
+ // table unwrapped both words and the operand looked executed.
817
+ if (inspectsOnly(argv))
818
+ return;
819
+ // The wrapper table keys on the BASENAME head (`/usr/bin/env` is `env`), so
820
+ // detection runs on a basenamed copy while the returned words stay verbatim.
821
+ // Wrappers only ever drop words off the FRONT, so a count maps the result
822
+ // back onto the original spellings.
823
+ const probe = [normalizeHead(argv[0] ?? ""), ...argv.slice(1)];
824
+ const dropped = probe.length - stripWrappers(probe).argv.length;
825
+ out.push(argv.slice(dropped));
826
+ };
827
+ // Both return TRUE when control provably does not continue past this node in
828
+ // the ENCLOSING shell — the list stops there.
829
+ const stmts = (list) => {
830
+ for (const st of list ?? [])
831
+ if (descend(st))
832
+ return true;
833
+ return false;
834
+ };
835
+ const descend = (node) => {
836
+ if (!node)
837
+ return false;
838
+ switch (sh.syntax.NodeType(node)) {
839
+ case "Stmt":
840
+ // `cmd &` runs in a background SUBSHELL, so a terminator inside it never
841
+ // reaches this shell (measured: `exit 0 & ./x.sh` runs `./x.sh`).
842
+ return descend(node.Cmd) && node.Background !== true;
843
+ case "CallExpr":
844
+ emit(node);
845
+ return terminates(node);
846
+ case "BinaryCmd":
847
+ // `&&` (10) and `||` (11) short-circuit, so only X is certain; a
848
+ // PIPELINE (`|` 12, `|&` 13) runs both sides.
849
+ if (node.Op === PIPE_OP || node.Op === PIPE_ALL_OP) {
850
+ descend(node.X);
851
+ descend(node.Y);
852
+ // Each side of a pipeline is its own subshell — `exit 0 | cat; ./x.sh`
853
+ // runs `./x.sh` (measured).
854
+ return false;
855
+ }
856
+ return descend(node.X);
857
+ case "Subshell":
858
+ stmts(node.Stmts);
859
+ return false; // `( exit 0 ); ./x.sh` runs `./x.sh` (measured)
860
+ case "Block":
861
+ return stmts(node.Stmts); // `{ exit 0; }; ./x.sh` does NOT (measured)
862
+ case "FuncDecl":
863
+ // A declaration executes nothing, so it neither contributes leaves nor
864
+ // terminates: `f() { exit 0; }; ./x.sh` runs `./x.sh` (measured).
865
+ return false;
866
+ default:
867
+ // A conditional or deferred body — not entered, because whether it ran
868
+ // is a runtime fact this parse cannot have. But whether control REACHES
869
+ // the next statement is a separate question, and if the body can
870
+ // terminate the shell the answer is unknown, so the list stops here.
871
+ return mayTerminate(node);
872
+ }
873
+ };
874
+ stmts(file.Stmts);
875
+ return out;
876
+ }
877
+ /**
878
+ * Does this simple command end the shell, so that nothing after it in the same
879
+ * list runs? MEASURED against bash 5.2 and dash on 2026-08-12 — the numbers and
880
+ * the disagreement below are why this is not read off a POSIX table:
881
+ *
882
+ * ```
883
+ * bash dash
884
+ * exit 0; ./x.sh x NOT run x NOT run
885
+ * return 0; ./x.sh x RAN (+error) x NOT run
886
+ * exec ./x.sh; echo AFTER AFTER not run AFTER not run
887
+ * exec > /dev/null; ./x.sh x RAN x RAN
888
+ * ```
889
+ *
890
+ * - `exit` — unconditional, both shells.
891
+ * - `return` — the two shells DISAGREE at top level: bash prints "can only
892
+ * `return' from a function or sourced script" and carries on; dash stops.
893
+ * `leafArgvSource` never enters a `FuncDecl`, so every `return` it can see is
894
+ * one of those top-level ones. Where the shells disagree the rule is to
895
+ * abstain, and for a coverage probe abstaining means NOT crediting what
896
+ * follows — so it truncates. Under bash that under-credits by one line; the
897
+ * other choice would be a false grant under `/bin/sh`, which is the shell
898
+ * `spawnSync(..., { shell: true })` actually uses.
899
+ * - `exec` — only with a command. `exec > file` (redirections and nothing else)
900
+ * just rewires the current shell and execution continues; that is the neighbour
901
+ * this rule would most easily get wrong, and the measurement above is why it
902
+ * does not. The exec'd program itself is emitted as a leaf like any other.
903
+ */
904
+ function terminates(call) {
905
+ const head = call.Args?.[0] ? getLiteral(call.Args[0]) : null;
906
+ if (head === "exit" || head === "return")
907
+ return true;
908
+ // `exec` with at least one more word. A word that is only an option
909
+ // (`exec -c`) counts too: mistaking it for a terminator drops later leaves,
910
+ // which is the silent direction.
911
+ return head === "exec" && (call.Args?.length ?? 0) > 1;
912
+ }
913
+ /**
914
+ * Could this un-entered construct end the shell? A conservative YES stops the
915
+ * statement list, because `if [ -z "$X" ]; then exit 1; fi; bash hooks/x.sh` does
916
+ * not necessarily reach the hook (measured: with the branch taken, `./x.sh` does
917
+ * NOT run).
918
+ *
919
+ * Deliberately conservative in the direction of SILENCE: an `exit` that could
920
+ * only ever run inside a nested subshell still stops the list, costing one
921
+ * coverage line. The common guard shape is untouched — a conditional containing
922
+ * no terminator answers `false`, so `if …; then echo warn; fi; bash hooks/x.sh`
923
+ * still attributes the hook.
924
+ */
925
+ function mayTerminate(node) {
926
+ let found = false;
927
+ sh.syntax.Walk(node, (n) => {
928
+ if (found)
929
+ return false;
930
+ // A function BODY is not executed where it is written, so a `return`/`exit`
931
+ // inside one says nothing about control here.
932
+ if (sh.syntax.NodeType(n) === "FuncDecl")
933
+ return false;
934
+ if (sh.syntax.NodeType(n) === "CallExpr" && terminates(n))
935
+ found = true;
936
+ return !found;
937
+ });
938
+ return found;
939
+ }
940
+ /**
941
+ * Is this leaf an INSPECTION rather than an execution? Then it contributes no
942
+ * executed program, however script-shaped its operand.
943
+ *
944
+ * Same family as the interpreter's parse-only flags (`bash -n`, `node --check`):
945
+ * a word that turns "run this" into "tell me about this". Bash 5.2's own
946
+ * `help command`: *"Execute a simple command or display information about
947
+ * commands."*
948
+ *
949
+ * MEASURED, bash 5.2, with a marker file the script touches — `ran` is whether
950
+ * the operand actually executed:
951
+ *
952
+ * ```
953
+ * command -v ./pre.sh ran=no prints ./pre.sh
954
+ * command -V ./pre.sh ran=no prints "./pre.sh is ./pre.sh"
955
+ * command -pv ./pre.sh ran=no
956
+ * command -vp ./pre.sh ran=no
957
+ * command -Vp ./pre.sh ran=no
958
+ * command -p ./pre.sh ran=YES ← -p is a PATH choice, not an inspection
959
+ * command ./pre.sh ran=YES
960
+ * ```
961
+ *
962
+ * ⚠️ THE REST OF THE WRAPPER TABLE WAS ASKED THE SAME QUESTION, and `command` is
963
+ * the only member with an inspect-only mode. Run against this build, the
964
+ * neighbours the finding names attribute NOTHING ALREADY — not by a rule, but
965
+ * because they are not wrappers at all, so their operand is never reached:
966
+ *
967
+ * ```
968
+ * commandRefs("type hooks/pre.sh") → [] (ran=no)
969
+ * commandRefs("which hooks/pre.sh") → [] (ran=no)
970
+ * commandRefs("hash hooks/pre.sh") → [] (ran=no)
971
+ * ```
972
+ *
973
+ * `sudo -v` (validate) and `sudo -l` (list) take no command operand, so there is
974
+ * nothing to attribute; `xargs -t` PRINTS what it runs and then runs it, so it
975
+ * is not an inspection. Deliberately left out: `env`, `nice`, `timeout`,
976
+ * `nohup` — none has an inspect mode (`--help`/`--version` exit without an
977
+ * operand, so they cannot misattribute one).
978
+ *
979
+ * ⚠️ SCOPED TO THE COVERAGE EXTRACTOR ON PURPOSE. `leafArgvSource` has exactly
980
+ * one caller and it is attribution; the SAFETY extractor (`leafCommands`, a
981
+ * blanket walk) is untouched, keeping the opposite default established when the
982
+ * statement-list truncation landed.
983
+ */
984
+ function inspectsOnly(argv) {
985
+ if (normalizeHead(argv[0] ?? "") !== "command")
986
+ return false;
987
+ for (const w of argv.slice(1)) {
988
+ if (!w.startsWith("-") || w === "-")
989
+ return false; // reached the operand
990
+ if (w === "--")
991
+ return false;
992
+ if (/^-[pvV]+$/.test(w) && /[vV]/.test(w))
993
+ return true;
994
+ }
995
+ return false;
996
+ }
997
+ /** `BinaryCmd.Op` for `|` and `|&` — the two that run BOTH sides. */
998
+ const PIPE_OP = 12;
999
+ const PIPE_ALL_OP = 13;
722
1000
  /** Normalize a Stmt's redirections into {@link LeafRedirect}s (source order). */
723
1001
  function normalizeRedirects(redirs) {
724
1002
  return redirs.map((r) => {
@@ -0,0 +1,185 @@
1
+ /**
2
+ * Refuse to spend model calls when somebody else's test runner owns this process.
3
+ *
4
+ * ## The failure this removes
5
+ *
6
+ * A vigiles eval spawns `claude` and spends real money. It is meant to run on a
7
+ * schedule, from `vigiles eval`. But an author who names one `foo.test.mjs` — a
8
+ * name vigiles used to CREDIT as coverage — hands it to every default `npx vitest`
9
+ * / `npx jest` run in the repository, because that name matches their default
10
+ * patterns exactly (read out of the installed packages):
11
+ *
12
+ * vitest ** /*.{test,spec}.?(c|m)[jt]s?(x)
13
+ * jest ** /?(*.)+(spec|test).?([mc])[jt]s?(x)
14
+ * ** /__tests__/** /*.?([mc])[jt]s?(x) ← everything in that directory
15
+ *
16
+ * Measured 2026-08-11: a bare project with `.claude/skills/foo/foo.test.mjs` and a
17
+ * plain `npx vitest run` — it executed. "It lives under a dot-directory, nothing
18
+ * else will find it" is false. The bill arrives silently, on every push, in CI.
19
+ *
20
+ * The naming half is fixed elsewhere (those globs no longer count). This is the
21
+ * part that does not depend on anybody obeying a convention: the money cannot be
22
+ * spent from inside a foreign runner, whatever the file is called.
23
+ *
24
+ * ## 🪦 THE STATIC HALF WAS DELETED 2026-08-12, AND THIS IS NOW THE WHOLE DEFENCE
25
+ *
26
+ * There used to be a companion, `core/foreign-runner-tests.ts`: a lexical gate
27
+ * that read every file under a harness surface dir, decided FROM THE TEXT whether
28
+ * it drives an agent (`runEval(`, `runHarnessTest(`, …), and warned in the audit
29
+ * report when a default vitest/jest glob would collect it. It is gone, with its
30
+ * test file and the disk walker in `scan.ts` that fed it.
31
+ *
32
+ * It was deleted under a rule that file pre-registered, not on a whim:
33
+ *
34
+ * > A seventh false positive is not another bug to fix — it is the measurement
35
+ * > saying the lexical rule cannot be made tight enough, and the answer then is
36
+ * > to DELETE the gate and let the money guard be the whole defence.
37
+ *
38
+ * THE SEVENTH ARRIVED: `function testDriver(runEval) { runEval(fake); }`. The
39
+ * binding check looked at declarations and import aliases but never at PARAMETER
40
+ * lists, so an ordinary offline unit test injecting a fake was reported as
41
+ * spawning a real agent. Run, not argued — the gate over one file of that shape
42
+ * returned one finding, whose warning read *"It calls `runEval`, which spawns an
43
+ * agent"* and told the author to rename the file. That sentence was false about
44
+ * that file.
45
+ *
46
+ * Six shapes had already been narrowed away one at a time, and this one had been
47
+ * SELF-DISCLOSED in the file as a known residual ("costs one warning and needs a
48
+ * scope analysis this module cannot have"). Pre-disclosure is not a defence: a
49
+ * pre-registered stopping rule exists precisely to stop "we know, it is
50
+ * acceptable" from running forever, so the count was honoured rather than
51
+ * reasoned around.
52
+ *
53
+ * The measurement agreed independently. Across three corpora (this repo, a
54
+ * 43-harness consumer repo, five vendored third-party plugin repos — 3 613 js/ts
55
+ * files), and re-measured at deletion time over this repo's 941 files:
56
+ *
57
+ * true positives, ever 0
58
+ * distinct false-positive shapes reported 7
59
+ *
60
+ * Its only demonstrated effect on a real repository was an effect of NOT firing.
61
+ * Every error it made was the expensive kind: the remedy it printed is "rename
62
+ * this file", so a false warning costs someone a working test, while a missed
63
+ * warning costs nothing — the refusal below already fires at all four spawn doors
64
+ * (`judge.ts`, `eval.ts`, `scan-behavioral.ts`, `adapters/codex/eval.ts`), so a
65
+ * miss is at most a less friendly notice, never a bill.
66
+ *
67
+ * ⚠️ WHAT IS GENUINELY LOST: the early, report-time heads-up. An author who names
68
+ * a harness test `foo.test.mjs` now learns it at the refusal below rather than in
69
+ * the audit. That is the trade, made knowingly — an advisory that has never once
70
+ * been right is not worth a warning that has been wrong seven times.
71
+ *
72
+ * The removal is PINNED, not merely done: `scan-vendor.test.ts` still asserts that
73
+ * no `COLLECTS AND EXECUTES` warning appears on the vendored corpus, so bringing
74
+ * the gate back fails a test instead of passing silently.
75
+ *
76
+ * ## Why not an environment variable
77
+ *
78
+ * The obvious version is `process.env.VITEST`. It was proposed and rejected: that
79
+ * variable can sit in a `.env` or in a CI environment for unrelated reasons, and
80
+ * then a legitimate scheduled eval refuses to run — a guard that fires on correct
81
+ * input is a guard people delete.
82
+ *
83
+ * `process.argv[1]` is the path node was actually started with, which no stray
84
+ * configuration can forge. Measured, same spike:
85
+ *
86
+ * under `npx vitest run` → …/node_modules/vitest/dist/workers/forks.js
87
+ * under `node foo.eval.mjs`→ …/foo.eval.mjs
88
+ * under `vigiles eval` → the script vigiles spawned
89
+ *
90
+ * So the check is a POSITIVE identification of a known runner, not "am I the entry
91
+ * point". Positive is the conservative direction here: an unrecognised wrapper
92
+ * runs (and may cost money) rather than a legitimate run being blocked by a
93
+ * pattern nobody predicted.
94
+ *
95
+ * ## 🔴 `node --test` HAS NO argv SIGNAL AT ALL, and the line that claimed one
96
+ * was dead from the day it was written
97
+ *
98
+ * `["node --test", ["node_modules/.bin/node--test"]]` used to sit in the table
99
+ * below. It was added by analogy with the npx-installed runners and never
100
+ * measured. There is no such binary: Node's test runner is a FLAG on node
101
+ * itself, so nothing under `node_modules/.bin/` can appear in `argv[1]`, and the
102
+ * entry could not fire under any invocation. That is worse than its absence — it
103
+ * READ as coverage while the paid tier stayed reachable from a legacy harness
104
+ * named `*.test.mjs`, a name Node's own default patterns collect.
105
+ *
106
+ * Measured 2026-08-12 (Node 22.22, a fixture printing its own process facts):
107
+ *
108
+ * node --test foo.test.mjs
109
+ * argv[1] = /abs/foo.test.mjs ← the TEST FILE, no runner in sight
110
+ * execArgv = [] ← the flag is not here either
111
+ * NODE_TEST_CONTEXT = child-v8 ← the only signal
112
+ *
113
+ * node --test --experimental-test-isolation=none foo.test.mjs
114
+ * argv[1] = foo.test.mjs
115
+ * execArgv = ["--test", "--experimental-test-isolation=none"]
116
+ * NODE_TEST_CONTEXT = undefined ← in-process: no child, no var
117
+ *
118
+ * The two modes therefore need two different facts, and both are read here.
119
+ *
120
+ * ⚠️ THIS IS NOT THE `process.env.VITEST` IDIOM REJECTED ABOVE, and the
121
+ * difference is the reason it is allowed. `VITEST` is a USER-VISIBLE CONVENTION:
122
+ * a person can put it in a `.env` or a CI job for unrelated reasons, so a guard
123
+ * reading it fires on correct input. `NODE_TEST_CONTEXT` is set by NODE ITSELF in
124
+ * the child it spawns — the internal protocol between runner and test process,
125
+ * which nobody writes into their own environment. Same for `execArgv`: the flag
126
+ * list node was launched with (`--test` is explicitly refused inside
127
+ * `NODE_OPTIONS` — verified — so it cannot arrive from a stray env either). Do
128
+ * not "fix" this back into an argv fragment: for `node --test` an argv signal
129
+ * does not exist, so the choice is these facts or no detection at all.
130
+ *
131
+ * Pure: facts in, a name or null out. No process access, no filesystem.
132
+ */
133
+ /**
134
+ * The process facts identifying Node's own test runner, which leaves no trace in
135
+ * `argv[1]`. Passed in rather than read, so the decision stays pure.
136
+ */
137
+ export interface NodeTestFacts {
138
+ /** `process.execArgv` — carries `--test` when the runner is IN-PROCESS. */
139
+ readonly execArgv?: readonly string[];
140
+ /** `process.env.NODE_TEST_CONTEXT` — node sets it in the test CHILD it spawns. */
141
+ readonly nodeTestContext?: string | undefined;
142
+ }
143
+ /**
144
+ * The foreign test runner that started this process, or `null`.
145
+ *
146
+ * `argv1` is `process.argv[1]`; `node` carries the two facts that identify Node's
147
+ * built-in runner (see the header). Both are passed in rather than read, so the
148
+ * decision is a pure function the tests drive with values measured from real runs.
149
+ */
150
+ export declare function foreignRunner(argv1: string | undefined, node?: NodeTestFacts): string | null;
151
+ /**
152
+ * The refusal message. Separate from the check so a test can assert the WORDS —
153
+ * a guard that stops a run without saying what to do instead is a support ticket.
154
+ */
155
+ export declare function foreignRunnerRefusal(runner: string, what: string): string;
156
+ /**
157
+ * Refuse to spend model budget when a foreign test runner started this process.
158
+ *
159
+ * 🔴 CALL THIS AT EVERY REAL-MODEL SPAWN. It used to live privately in `eval.ts`
160
+ * and be called from exactly one place — `spawnAgent`, the real `claude` runner —
161
+ * on the reasoning that the composition root funnels every paid path. Measured
162
+ * 2026-08-12: it funnels ONE of four.
163
+ *
164
+ * src/eval.ts spawn(claudeCodeRuntime.agentBinary) guarded
165
+ * src/adapters/codex/eval.ts spawnSync("codex", …) UNGUARDED
166
+ * src/judge.ts spawnSync("claude", …) UNGUARDED
167
+ * src/scan-behavioral.ts spawnSync("claude", …) UNGUARDED
168
+ *
169
+ * `measureTriggerRate(spec, { evalDriver })` calls the injected driver's runner
170
+ * DIRECTLY, so a Codex eval collected by a stray `npx vitest run` spent real
171
+ * money on every push while the guard looked complete. The irony is worth
172
+ * recording: the untested-skill nudge now tells Codex users to pass
173
+ * `{ evalDriver }` — the product recommends the shape that bypassed the guard.
174
+ *
175
+ * Lives here, in a module that imports NOTHING, so any adapter can call it
176
+ * without a cycle. `foreignRunner` stays pure; this is the one impure wrapper,
177
+ * and it is impure precisely so callers cannot forget to pass the facts.
178
+ *
179
+ * ⚠️ CALL IT OUTSIDE ANY `try` THAT SWALLOWS. `judge` and `deriveAttackReal` both
180
+ * wrap their spawn in `try { … } catch { return fallback }`, so a refusal thrown
181
+ * inside would be caught and downgraded to a score of 0 / a canned string — a
182
+ * silent wrong answer instead of a stop. Both call it before the `try`.
183
+ */
184
+ export declare function refuseUnderForeignRunner(what: string): void;
185
+ //# sourceMappingURL=foreign-runner.d.ts.map