vigiles 15.0.2 → 15.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/layout.js +3 -0
- package/dist/adapters/claude-code/run-scripts.d.ts +21 -9
- package/dist/adapters/claude-code/run-scripts.js +33 -22
- package/dist/adapters/codex/driver.js +3 -1
- package/dist/adapters/codex/eval.js +5 -0
- package/dist/audit-score.js +32 -6
- package/dist/check-count.d.ts +62 -3
- package/dist/check-count.js +123 -5
- package/dist/check.d.ts +55 -0
- package/dist/check.js +91 -0
- package/dist/cli.js +456 -54
- package/dist/core/bash-effects.d.ts +41 -0
- package/dist/core/bash-effects.js +278 -0
- package/dist/core/foreign-runner.d.ts +185 -0
- package/dist/core/foreign-runner.js +228 -0
- package/dist/core/harness-driver.d.ts +18 -1
- package/dist/core/hook-program.d.ts +319 -20
- package/dist/core/hook-program.js +743 -49
- package/dist/core/layout.d.ts +13 -0
- package/dist/core/lethal-trifecta.d.ts +93 -0
- package/dist/core/lethal-trifecta.js +409 -65
- package/dist/core/markdown.d.ts +32 -0
- package/dist/core/markdown.js +36 -0
- package/dist/core/merge-conflict.d.ts +50 -0
- package/dist/core/merge-conflict.js +84 -0
- package/dist/core/test-file-ext.d.ts +80 -0
- package/dist/core/test-file-ext.js +90 -0
- package/dist/core/types.d.ts +11 -0
- package/dist/coverage-artifact.d.ts +383 -0
- package/dist/coverage-artifact.js +586 -0
- package/dist/coverage-evidence.d.ts +87 -79
- package/dist/coverage-evidence.js +212 -211
- package/dist/coverage-probe.d.ts +125 -0
- package/dist/coverage-probe.js +700 -0
- package/dist/doc-commands.d.ts +119 -0
- package/dist/doc-commands.js +158 -0
- package/dist/eval-lock.d.ts +9 -0
- package/dist/eval-lock.js +14 -2
- package/dist/eval.js +28 -0
- package/dist/fs-walk.d.ts +80 -0
- package/dist/fs-walk.js +148 -0
- package/dist/harness-assert.js +42 -3
- package/dist/harness-test.js +31 -2
- package/dist/judge.js +4 -0
- package/dist/leaderboard.d.ts +16 -1
- package/dist/leaderboard.js +19 -2
- package/dist/load-hook.js +6 -1
- package/dist/mock-model.d.ts +58 -4
- package/dist/mock-model.js +133 -5
- package/dist/observe.d.ts +1 -1
- package/dist/observe.js +44 -1
- package/dist/plugin-loader.js +52 -13
- package/dist/run-hook.d.ts +58 -0
- package/dist/run-hook.js +70 -0
- package/dist/run-script.js +75 -0
- package/dist/scaffold-test.d.ts +3 -0
- package/dist/scaffold-test.js +10 -5
- package/dist/scan-behavioral.js +4 -0
- package/dist/scan-core.d.ts +22 -0
- package/dist/scan-core.js +46 -14
- package/dist/scan-files.js +19 -2
- package/dist/scan.d.ts +22 -5
- package/dist/scan.js +67 -13
- package/dist/skill-contract.d.ts +46 -0
- package/dist/skill-contract.js +201 -0
- package/dist/skill-refs.d.ts +67 -0
- package/dist/skill-refs.js +117 -0
- package/dist/test-coverage-files.d.ts +7 -1
- package/dist/test-coverage-files.js +66 -55
- package/dist/test-coverage.d.ts +155 -19
- package/dist/test-coverage.js +430 -91
- package/dist/testing.d.ts +8 -2
- package/dist/testing.js +19 -1
- package/dist/trigger-containment.d.ts +85 -0
- package/dist/trigger-containment.js +125 -0
- package/dist/ts-runner-caps.d.ts +16 -0
- package/dist/ts-runner-caps.js +43 -0
- package/dist/unit.d.ts +2 -2
- package/dist/unit.js +2 -1
- package/hooks/eval-lock-nudge.sh +14 -6
- package/package.json +2 -2
- package/skills/test-harness/SKILL.md +68 -2
|
@@ -131,4 +131,45 @@ export interface NormalizedLeaf {
|
|
|
131
131
|
* failure → []. A leaf with a dynamic head is skipped (can't be normalized).
|
|
132
132
|
*/
|
|
133
133
|
export declare function leafCommandsNormalized(command: string): NormalizedLeaf[];
|
|
134
|
+
/**
|
|
135
|
+
* Every simple command's argv, POSITIONALLY, each word reconstructed at source
|
|
136
|
+
* level — the primitive a caller needs to tell an EXECUTED PROGRAM from a DATA
|
|
137
|
+
* OPERAND.
|
|
138
|
+
*
|
|
139
|
+
* 🔴 WHY POSITION, AND WHY A THIRD EXTRACTOR. `leafCommands` DROPS the words it
|
|
140
|
+
* cannot reduce to a literal, so `"$GUARD" --flag` yields `["--flag"]` — a
|
|
141
|
+
* dynamic head silently promotes an argument into head position, which is worse
|
|
142
|
+
* than useless to a positional reader. `leafCommandsNormalized` skips such a leaf
|
|
143
|
+
* outright AND basenames the head, so `./hooks/x.sh` loses the path a file
|
|
144
|
+
* resolver needs. This one keeps every word in its slot (an unreconstructable
|
|
145
|
+
* word becomes `""`) and keeps the head's spelling.
|
|
146
|
+
*
|
|
147
|
+
* Wrappers are still resolved through (`env FOO=1 bash x.sh` → `bash x.sh`)
|
|
148
|
+
* using the same wrapper table as the normalized extractor, not a second copy.
|
|
149
|
+
*
|
|
150
|
+
* 🔴 ONLY THE LEAVES THAT UNCONDITIONALLY RUN, and this used to be a blanket
|
|
151
|
+
* `Walk` over every `CallExpr` in the tree. A syntactic leaf is not an executed
|
|
152
|
+
* command: `false && bash hooks/pre.sh` and `true || bash hooks/pre.sh` never run
|
|
153
|
+
* the hook, and `if false; then bash hooks/x.sh; fi` and a function BODY do not
|
|
154
|
+
* run at all where they are written — yet all four were reported as executed
|
|
155
|
+
* programs. For the one caller (coverage attribution) that is a FALSE GRANT: a
|
|
156
|
+
* hook credited with a run that never happened. Same class as attributing a data
|
|
157
|
+
* operand, one level up in the grammar.
|
|
158
|
+
*
|
|
159
|
+
* So the tree is DESCENDED, not walked, and only through positions whose children
|
|
160
|
+
* always execute: a statement list, both sides of a PIPELINE, a subshell, a
|
|
161
|
+
* block. `&&`/`||` contribute their LEFT side only — the right is conditional.
|
|
162
|
+
* Every other construct (`if`, `while`, `until`, `for`, `case`, a function
|
|
163
|
+
* declaration, `time`, …) is not entered at all: whether its body ran is a
|
|
164
|
+
* runtime fact this parse cannot have, and abstaining costs one warning while
|
|
165
|
+
* guessing costs a false claim that something was tested.
|
|
166
|
+
*
|
|
167
|
+
* ⚠️ THE MEASURED COST, stated rather than assumed: `cd /repo && bash hooks/x.sh`
|
|
168
|
+
* now yields NOTHING, because the hook sits on the right of `&&`. That idiom has
|
|
169
|
+
* a first-class replacement — `runHook(cmd, event, { cwd })` — which is how this
|
|
170
|
+
* repo's own examples already write it.
|
|
171
|
+
*
|
|
172
|
+
* Parse failure → `[]`.
|
|
173
|
+
*/
|
|
174
|
+
export declare function leafArgvSource(command: string): string[][];
|
|
134
175
|
//# sourceMappingURL=bash-effects.d.ts.map
|
|
@@ -28,6 +28,7 @@ exports.classifyBashCommand = classifyBashCommand;
|
|
|
28
28
|
exports.isReadOnlyBash = isReadOnlyBash;
|
|
29
29
|
exports.leafCommands = leafCommands;
|
|
30
30
|
exports.leafCommandsNormalized = leafCommandsNormalized;
|
|
31
|
+
exports.leafArgvSource = leafArgvSource;
|
|
31
32
|
// mvdan-sh is a CJS package (GopherJS build) with no bundled TypeScript types.
|
|
32
33
|
// The project compiles to CommonJS (Node16, no "type":"module"), so plain
|
|
33
34
|
// require() works and is the idiomatic pattern here (see linters.ts).
|
|
@@ -719,6 +720,283 @@ function leafCommandsNormalized(command) {
|
|
|
719
720
|
});
|
|
720
721
|
return out;
|
|
721
722
|
}
|
|
723
|
+
/**
|
|
724
|
+
* Reconstruct a Word's SOURCE-level text: quotes unwrapped, parameter references
|
|
725
|
+
* kept verbatim (`${CLAUDE_PROJECT_DIR}` → `$CLAUDE_PROJECT_DIR`). `null` when a
|
|
726
|
+
* segment cannot be reconstructed at all (command substitution, arithmetic,
|
|
727
|
+
* process substitution).
|
|
728
|
+
*
|
|
729
|
+
* The twin of {@link normalizeParts}, and deliberately NOT the same function.
|
|
730
|
+
* `normalizeParts` answers "what OPERATION is this" and therefore collapses
|
|
731
|
+
* `$HOME` to `~` and gives up on every other parameter; a caller that needs the
|
|
732
|
+
* PATH a word names can use neither behaviour — `$CLAUDE_PROJECT_DIR/.claude/hooks/x.sh`
|
|
733
|
+
* has to survive as written, because a file resolver matches it by suffix.
|
|
734
|
+
*/
|
|
735
|
+
function sourceParts(parts) {
|
|
736
|
+
if (!parts)
|
|
737
|
+
return null;
|
|
738
|
+
let out = "";
|
|
739
|
+
for (const p of parts) {
|
|
740
|
+
const t = sh.syntax.NodeType(p);
|
|
741
|
+
if (t === "Lit" || t === "SglQuoted") {
|
|
742
|
+
out += p.Value ?? "";
|
|
743
|
+
}
|
|
744
|
+
else if (t === "DblQuoted") {
|
|
745
|
+
const inner = sourceParts(p.Parts);
|
|
746
|
+
if (inner === null)
|
|
747
|
+
return null;
|
|
748
|
+
out += inner;
|
|
749
|
+
}
|
|
750
|
+
else if (t === "ParamExp") {
|
|
751
|
+
const name = p.Param?.Value;
|
|
752
|
+
if (!name)
|
|
753
|
+
return null;
|
|
754
|
+
out += `$${name}`;
|
|
755
|
+
}
|
|
756
|
+
else {
|
|
757
|
+
return null; // CmdSubst / ArithmExp / ProcSubst / … → unreconstructable
|
|
758
|
+
}
|
|
759
|
+
}
|
|
760
|
+
return out;
|
|
761
|
+
}
|
|
762
|
+
/**
|
|
763
|
+
* Every simple command's argv, POSITIONALLY, each word reconstructed at source
|
|
764
|
+
* level — the primitive a caller needs to tell an EXECUTED PROGRAM from a DATA
|
|
765
|
+
* OPERAND.
|
|
766
|
+
*
|
|
767
|
+
* 🔴 WHY POSITION, AND WHY A THIRD EXTRACTOR. `leafCommands` DROPS the words it
|
|
768
|
+
* cannot reduce to a literal, so `"$GUARD" --flag` yields `["--flag"]` — a
|
|
769
|
+
* dynamic head silently promotes an argument into head position, which is worse
|
|
770
|
+
* than useless to a positional reader. `leafCommandsNormalized` skips such a leaf
|
|
771
|
+
* outright AND basenames the head, so `./hooks/x.sh` loses the path a file
|
|
772
|
+
* resolver needs. This one keeps every word in its slot (an unreconstructable
|
|
773
|
+
* word becomes `""`) and keeps the head's spelling.
|
|
774
|
+
*
|
|
775
|
+
* Wrappers are still resolved through (`env FOO=1 bash x.sh` → `bash x.sh`)
|
|
776
|
+
* using the same wrapper table as the normalized extractor, not a second copy.
|
|
777
|
+
*
|
|
778
|
+
* 🔴 ONLY THE LEAVES THAT UNCONDITIONALLY RUN, and this used to be a blanket
|
|
779
|
+
* `Walk` over every `CallExpr` in the tree. A syntactic leaf is not an executed
|
|
780
|
+
* command: `false && bash hooks/pre.sh` and `true || bash hooks/pre.sh` never run
|
|
781
|
+
* the hook, and `if false; then bash hooks/x.sh; fi` and a function BODY do not
|
|
782
|
+
* run at all where they are written — yet all four were reported as executed
|
|
783
|
+
* programs. For the one caller (coverage attribution) that is a FALSE GRANT: a
|
|
784
|
+
* hook credited with a run that never happened. Same class as attributing a data
|
|
785
|
+
* operand, one level up in the grammar.
|
|
786
|
+
*
|
|
787
|
+
* So the tree is DESCENDED, not walked, and only through positions whose children
|
|
788
|
+
* always execute: a statement list, both sides of a PIPELINE, a subshell, a
|
|
789
|
+
* block. `&&`/`||` contribute their LEFT side only — the right is conditional.
|
|
790
|
+
* Every other construct (`if`, `while`, `until`, `for`, `case`, a function
|
|
791
|
+
* declaration, `time`, …) is not entered at all: whether its body ran is a
|
|
792
|
+
* runtime fact this parse cannot have, and abstaining costs one warning while
|
|
793
|
+
* guessing costs a false claim that something was tested.
|
|
794
|
+
*
|
|
795
|
+
* ⚠️ THE MEASURED COST, stated rather than assumed: `cd /repo && bash hooks/x.sh`
|
|
796
|
+
* now yields NOTHING, because the hook sits on the right of `&&`. That idiom has
|
|
797
|
+
* a first-class replacement — `runHook(cmd, event, { cwd })` — which is how this
|
|
798
|
+
* repo's own examples already write it.
|
|
799
|
+
*
|
|
800
|
+
* Parse failure → `[]`.
|
|
801
|
+
*/
|
|
802
|
+
function leafArgvSource(command) {
|
|
803
|
+
let file;
|
|
804
|
+
try {
|
|
805
|
+
file = sh.syntax.NewParser().Parse(command, "cmd.sh");
|
|
806
|
+
}
|
|
807
|
+
catch {
|
|
808
|
+
return [];
|
|
809
|
+
}
|
|
810
|
+
const out = [];
|
|
811
|
+
const emit = (node) => {
|
|
812
|
+
if (!node.Args?.length)
|
|
813
|
+
return;
|
|
814
|
+
const argv = node.Args.map((w) => sourceParts(w.Parts) ?? "");
|
|
815
|
+
// `command -v x` DESCRIBES x; it does not run it. Without this the wrapper
|
|
816
|
+
// table unwrapped both words and the operand looked executed.
|
|
817
|
+
if (inspectsOnly(argv))
|
|
818
|
+
return;
|
|
819
|
+
// The wrapper table keys on the BASENAME head (`/usr/bin/env` is `env`), so
|
|
820
|
+
// detection runs on a basenamed copy while the returned words stay verbatim.
|
|
821
|
+
// Wrappers only ever drop words off the FRONT, so a count maps the result
|
|
822
|
+
// back onto the original spellings.
|
|
823
|
+
const probe = [normalizeHead(argv[0] ?? ""), ...argv.slice(1)];
|
|
824
|
+
const dropped = probe.length - stripWrappers(probe).argv.length;
|
|
825
|
+
out.push(argv.slice(dropped));
|
|
826
|
+
};
|
|
827
|
+
// Both return TRUE when control provably does not continue past this node in
|
|
828
|
+
// the ENCLOSING shell — the list stops there.
|
|
829
|
+
const stmts = (list) => {
|
|
830
|
+
for (const st of list ?? [])
|
|
831
|
+
if (descend(st))
|
|
832
|
+
return true;
|
|
833
|
+
return false;
|
|
834
|
+
};
|
|
835
|
+
const descend = (node) => {
|
|
836
|
+
if (!node)
|
|
837
|
+
return false;
|
|
838
|
+
switch (sh.syntax.NodeType(node)) {
|
|
839
|
+
case "Stmt":
|
|
840
|
+
// `cmd &` runs in a background SUBSHELL, so a terminator inside it never
|
|
841
|
+
// reaches this shell (measured: `exit 0 & ./x.sh` runs `./x.sh`).
|
|
842
|
+
return descend(node.Cmd) && node.Background !== true;
|
|
843
|
+
case "CallExpr":
|
|
844
|
+
emit(node);
|
|
845
|
+
return terminates(node);
|
|
846
|
+
case "BinaryCmd":
|
|
847
|
+
// `&&` (10) and `||` (11) short-circuit, so only X is certain; a
|
|
848
|
+
// PIPELINE (`|` 12, `|&` 13) runs both sides.
|
|
849
|
+
if (node.Op === PIPE_OP || node.Op === PIPE_ALL_OP) {
|
|
850
|
+
descend(node.X);
|
|
851
|
+
descend(node.Y);
|
|
852
|
+
// Each side of a pipeline is its own subshell — `exit 0 | cat; ./x.sh`
|
|
853
|
+
// runs `./x.sh` (measured).
|
|
854
|
+
return false;
|
|
855
|
+
}
|
|
856
|
+
return descend(node.X);
|
|
857
|
+
case "Subshell":
|
|
858
|
+
stmts(node.Stmts);
|
|
859
|
+
return false; // `( exit 0 ); ./x.sh` runs `./x.sh` (measured)
|
|
860
|
+
case "Block":
|
|
861
|
+
return stmts(node.Stmts); // `{ exit 0; }; ./x.sh` does NOT (measured)
|
|
862
|
+
case "FuncDecl":
|
|
863
|
+
// A declaration executes nothing, so it neither contributes leaves nor
|
|
864
|
+
// terminates: `f() { exit 0; }; ./x.sh` runs `./x.sh` (measured).
|
|
865
|
+
return false;
|
|
866
|
+
default:
|
|
867
|
+
// A conditional or deferred body — not entered, because whether it ran
|
|
868
|
+
// is a runtime fact this parse cannot have. But whether control REACHES
|
|
869
|
+
// the next statement is a separate question, and if the body can
|
|
870
|
+
// terminate the shell the answer is unknown, so the list stops here.
|
|
871
|
+
return mayTerminate(node);
|
|
872
|
+
}
|
|
873
|
+
};
|
|
874
|
+
stmts(file.Stmts);
|
|
875
|
+
return out;
|
|
876
|
+
}
|
|
877
|
+
/**
|
|
878
|
+
* Does this simple command end the shell, so that nothing after it in the same
|
|
879
|
+
* list runs? MEASURED against bash 5.2 and dash on 2026-08-12 — the numbers and
|
|
880
|
+
* the disagreement below are why this is not read off a POSIX table:
|
|
881
|
+
*
|
|
882
|
+
* ```
|
|
883
|
+
* bash dash
|
|
884
|
+
* exit 0; ./x.sh x NOT run x NOT run
|
|
885
|
+
* return 0; ./x.sh x RAN (+error) x NOT run
|
|
886
|
+
* exec ./x.sh; echo AFTER AFTER not run AFTER not run
|
|
887
|
+
* exec > /dev/null; ./x.sh x RAN x RAN
|
|
888
|
+
* ```
|
|
889
|
+
*
|
|
890
|
+
* - `exit` — unconditional, both shells.
|
|
891
|
+
* - `return` — the two shells DISAGREE at top level: bash prints "can only
|
|
892
|
+
* `return' from a function or sourced script" and carries on; dash stops.
|
|
893
|
+
* `leafArgvSource` never enters a `FuncDecl`, so every `return` it can see is
|
|
894
|
+
* one of those top-level ones. Where the shells disagree the rule is to
|
|
895
|
+
* abstain, and for a coverage probe abstaining means NOT crediting what
|
|
896
|
+
* follows — so it truncates. Under bash that under-credits by one line; the
|
|
897
|
+
* other choice would be a false grant under `/bin/sh`, which is the shell
|
|
898
|
+
* `spawnSync(..., { shell: true })` actually uses.
|
|
899
|
+
* - `exec` — only with a command. `exec > file` (redirections and nothing else)
|
|
900
|
+
* just rewires the current shell and execution continues; that is the neighbour
|
|
901
|
+
* this rule would most easily get wrong, and the measurement above is why it
|
|
902
|
+
* does not. The exec'd program itself is emitted as a leaf like any other.
|
|
903
|
+
*/
|
|
904
|
+
function terminates(call) {
|
|
905
|
+
const head = call.Args?.[0] ? getLiteral(call.Args[0]) : null;
|
|
906
|
+
if (head === "exit" || head === "return")
|
|
907
|
+
return true;
|
|
908
|
+
// `exec` with at least one more word. A word that is only an option
|
|
909
|
+
// (`exec -c`) counts too: mistaking it for a terminator drops later leaves,
|
|
910
|
+
// which is the silent direction.
|
|
911
|
+
return head === "exec" && (call.Args?.length ?? 0) > 1;
|
|
912
|
+
}
|
|
913
|
+
/**
|
|
914
|
+
* Could this un-entered construct end the shell? A conservative YES stops the
|
|
915
|
+
* statement list, because `if [ -z "$X" ]; then exit 1; fi; bash hooks/x.sh` does
|
|
916
|
+
* not necessarily reach the hook (measured: with the branch taken, `./x.sh` does
|
|
917
|
+
* NOT run).
|
|
918
|
+
*
|
|
919
|
+
* Deliberately conservative in the direction of SILENCE: an `exit` that could
|
|
920
|
+
* only ever run inside a nested subshell still stops the list, costing one
|
|
921
|
+
* coverage line. The common guard shape is untouched — a conditional containing
|
|
922
|
+
* no terminator answers `false`, so `if …; then echo warn; fi; bash hooks/x.sh`
|
|
923
|
+
* still attributes the hook.
|
|
924
|
+
*/
|
|
925
|
+
function mayTerminate(node) {
|
|
926
|
+
let found = false;
|
|
927
|
+
sh.syntax.Walk(node, (n) => {
|
|
928
|
+
if (found)
|
|
929
|
+
return false;
|
|
930
|
+
// A function BODY is not executed where it is written, so a `return`/`exit`
|
|
931
|
+
// inside one says nothing about control here.
|
|
932
|
+
if (sh.syntax.NodeType(n) === "FuncDecl")
|
|
933
|
+
return false;
|
|
934
|
+
if (sh.syntax.NodeType(n) === "CallExpr" && terminates(n))
|
|
935
|
+
found = true;
|
|
936
|
+
return !found;
|
|
937
|
+
});
|
|
938
|
+
return found;
|
|
939
|
+
}
|
|
940
|
+
/**
|
|
941
|
+
* Is this leaf an INSPECTION rather than an execution? Then it contributes no
|
|
942
|
+
* executed program, however script-shaped its operand.
|
|
943
|
+
*
|
|
944
|
+
* Same family as the interpreter's parse-only flags (`bash -n`, `node --check`):
|
|
945
|
+
* a word that turns "run this" into "tell me about this". Bash 5.2's own
|
|
946
|
+
* `help command`: *"Execute a simple command or display information about
|
|
947
|
+
* commands."*
|
|
948
|
+
*
|
|
949
|
+
* MEASURED, bash 5.2, with a marker file the script touches — `ran` is whether
|
|
950
|
+
* the operand actually executed:
|
|
951
|
+
*
|
|
952
|
+
* ```
|
|
953
|
+
* command -v ./pre.sh ran=no prints ./pre.sh
|
|
954
|
+
* command -V ./pre.sh ran=no prints "./pre.sh is ./pre.sh"
|
|
955
|
+
* command -pv ./pre.sh ran=no
|
|
956
|
+
* command -vp ./pre.sh ran=no
|
|
957
|
+
* command -Vp ./pre.sh ran=no
|
|
958
|
+
* command -p ./pre.sh ran=YES ← -p is a PATH choice, not an inspection
|
|
959
|
+
* command ./pre.sh ran=YES
|
|
960
|
+
* ```
|
|
961
|
+
*
|
|
962
|
+
* ⚠️ THE REST OF THE WRAPPER TABLE WAS ASKED THE SAME QUESTION, and `command` is
|
|
963
|
+
* the only member with an inspect-only mode. Run against this build, the
|
|
964
|
+
* neighbours the finding names attribute NOTHING ALREADY — not by a rule, but
|
|
965
|
+
* because they are not wrappers at all, so their operand is never reached:
|
|
966
|
+
*
|
|
967
|
+
* ```
|
|
968
|
+
* commandRefs("type hooks/pre.sh") → [] (ran=no)
|
|
969
|
+
* commandRefs("which hooks/pre.sh") → [] (ran=no)
|
|
970
|
+
* commandRefs("hash hooks/pre.sh") → [] (ran=no)
|
|
971
|
+
* ```
|
|
972
|
+
*
|
|
973
|
+
* `sudo -v` (validate) and `sudo -l` (list) take no command operand, so there is
|
|
974
|
+
* nothing to attribute; `xargs -t` PRINTS what it runs and then runs it, so it
|
|
975
|
+
* is not an inspection. Deliberately left out: `env`, `nice`, `timeout`,
|
|
976
|
+
* `nohup` — none has an inspect mode (`--help`/`--version` exit without an
|
|
977
|
+
* operand, so they cannot misattribute one).
|
|
978
|
+
*
|
|
979
|
+
* ⚠️ SCOPED TO THE COVERAGE EXTRACTOR ON PURPOSE. `leafArgvSource` has exactly
|
|
980
|
+
* one caller and it is attribution; the SAFETY extractor (`leafCommands`, a
|
|
981
|
+
* blanket walk) is untouched, keeping the opposite default established when the
|
|
982
|
+
* statement-list truncation landed.
|
|
983
|
+
*/
|
|
984
|
+
function inspectsOnly(argv) {
|
|
985
|
+
if (normalizeHead(argv[0] ?? "") !== "command")
|
|
986
|
+
return false;
|
|
987
|
+
for (const w of argv.slice(1)) {
|
|
988
|
+
if (!w.startsWith("-") || w === "-")
|
|
989
|
+
return false; // reached the operand
|
|
990
|
+
if (w === "--")
|
|
991
|
+
return false;
|
|
992
|
+
if (/^-[pvV]+$/.test(w) && /[vV]/.test(w))
|
|
993
|
+
return true;
|
|
994
|
+
}
|
|
995
|
+
return false;
|
|
996
|
+
}
|
|
997
|
+
/** `BinaryCmd.Op` for `|` and `|&` — the two that run BOTH sides. */
|
|
998
|
+
const PIPE_OP = 12;
|
|
999
|
+
const PIPE_ALL_OP = 13;
|
|
722
1000
|
/** Normalize a Stmt's redirections into {@link LeafRedirect}s (source order). */
|
|
723
1001
|
function normalizeRedirects(redirs) {
|
|
724
1002
|
return redirs.map((r) => {
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Refuse to spend model calls when somebody else's test runner owns this process.
|
|
3
|
+
*
|
|
4
|
+
* ## The failure this removes
|
|
5
|
+
*
|
|
6
|
+
* A vigiles eval spawns `claude` and spends real money. It is meant to run on a
|
|
7
|
+
* schedule, from `vigiles eval`. But an author who names one `foo.test.mjs` — a
|
|
8
|
+
* name vigiles used to CREDIT as coverage — hands it to every default `npx vitest`
|
|
9
|
+
* / `npx jest` run in the repository, because that name matches their default
|
|
10
|
+
* patterns exactly (read out of the installed packages):
|
|
11
|
+
*
|
|
12
|
+
* vitest ** /*.{test,spec}.?(c|m)[jt]s?(x)
|
|
13
|
+
* jest ** /?(*.)+(spec|test).?([mc])[jt]s?(x)
|
|
14
|
+
* ** /__tests__/** /*.?([mc])[jt]s?(x) ← everything in that directory
|
|
15
|
+
*
|
|
16
|
+
* Measured 2026-08-11: a bare project with `.claude/skills/foo/foo.test.mjs` and a
|
|
17
|
+
* plain `npx vitest run` — it executed. "It lives under a dot-directory, nothing
|
|
18
|
+
* else will find it" is false. The bill arrives silently, on every push, in CI.
|
|
19
|
+
*
|
|
20
|
+
* The naming half is fixed elsewhere (those globs no longer count). This is the
|
|
21
|
+
* part that does not depend on anybody obeying a convention: the money cannot be
|
|
22
|
+
* spent from inside a foreign runner, whatever the file is called.
|
|
23
|
+
*
|
|
24
|
+
* ## 🪦 THE STATIC HALF WAS DELETED 2026-08-12, AND THIS IS NOW THE WHOLE DEFENCE
|
|
25
|
+
*
|
|
26
|
+
* There used to be a companion, `core/foreign-runner-tests.ts`: a lexical gate
|
|
27
|
+
* that read every file under a harness surface dir, decided FROM THE TEXT whether
|
|
28
|
+
* it drives an agent (`runEval(`, `runHarnessTest(`, …), and warned in the audit
|
|
29
|
+
* report when a default vitest/jest glob would collect it. It is gone, with its
|
|
30
|
+
* test file and the disk walker in `scan.ts` that fed it.
|
|
31
|
+
*
|
|
32
|
+
* It was deleted under a rule that file pre-registered, not on a whim:
|
|
33
|
+
*
|
|
34
|
+
* > A seventh false positive is not another bug to fix — it is the measurement
|
|
35
|
+
* > saying the lexical rule cannot be made tight enough, and the answer then is
|
|
36
|
+
* > to DELETE the gate and let the money guard be the whole defence.
|
|
37
|
+
*
|
|
38
|
+
* THE SEVENTH ARRIVED: `function testDriver(runEval) { runEval(fake); }`. The
|
|
39
|
+
* binding check looked at declarations and import aliases but never at PARAMETER
|
|
40
|
+
* lists, so an ordinary offline unit test injecting a fake was reported as
|
|
41
|
+
* spawning a real agent. Run, not argued — the gate over one file of that shape
|
|
42
|
+
* returned one finding, whose warning read *"It calls `runEval`, which spawns an
|
|
43
|
+
* agent"* and told the author to rename the file. That sentence was false about
|
|
44
|
+
* that file.
|
|
45
|
+
*
|
|
46
|
+
* Six shapes had already been narrowed away one at a time, and this one had been
|
|
47
|
+
* SELF-DISCLOSED in the file as a known residual ("costs one warning and needs a
|
|
48
|
+
* scope analysis this module cannot have"). Pre-disclosure is not a defence: a
|
|
49
|
+
* pre-registered stopping rule exists precisely to stop "we know, it is
|
|
50
|
+
* acceptable" from running forever, so the count was honoured rather than
|
|
51
|
+
* reasoned around.
|
|
52
|
+
*
|
|
53
|
+
* The measurement agreed independently. Across three corpora (this repo, a
|
|
54
|
+
* 43-harness consumer repo, five vendored third-party plugin repos — 3 613 js/ts
|
|
55
|
+
* files), and re-measured at deletion time over this repo's 941 files:
|
|
56
|
+
*
|
|
57
|
+
* true positives, ever 0
|
|
58
|
+
* distinct false-positive shapes reported 7
|
|
59
|
+
*
|
|
60
|
+
* Its only demonstrated effect on a real repository was an effect of NOT firing.
|
|
61
|
+
* Every error it made was the expensive kind: the remedy it printed is "rename
|
|
62
|
+
* this file", so a false warning costs someone a working test, while a missed
|
|
63
|
+
* warning costs nothing — the refusal below already fires at all four spawn doors
|
|
64
|
+
* (`judge.ts`, `eval.ts`, `scan-behavioral.ts`, `adapters/codex/eval.ts`), so a
|
|
65
|
+
* miss is at most a less friendly notice, never a bill.
|
|
66
|
+
*
|
|
67
|
+
* ⚠️ WHAT IS GENUINELY LOST: the early, report-time heads-up. An author who names
|
|
68
|
+
* a harness test `foo.test.mjs` now learns it at the refusal below rather than in
|
|
69
|
+
* the audit. That is the trade, made knowingly — an advisory that has never once
|
|
70
|
+
* been right is not worth a warning that has been wrong seven times.
|
|
71
|
+
*
|
|
72
|
+
* The removal is PINNED, not merely done: `scan-vendor.test.ts` still asserts that
|
|
73
|
+
* no `COLLECTS AND EXECUTES` warning appears on the vendored corpus, so bringing
|
|
74
|
+
* the gate back fails a test instead of passing silently.
|
|
75
|
+
*
|
|
76
|
+
* ## Why not an environment variable
|
|
77
|
+
*
|
|
78
|
+
* The obvious version is `process.env.VITEST`. It was proposed and rejected: that
|
|
79
|
+
* variable can sit in a `.env` or in a CI environment for unrelated reasons, and
|
|
80
|
+
* then a legitimate scheduled eval refuses to run — a guard that fires on correct
|
|
81
|
+
* input is a guard people delete.
|
|
82
|
+
*
|
|
83
|
+
* `process.argv[1]` is the path node was actually started with, which no stray
|
|
84
|
+
* configuration can forge. Measured, same spike:
|
|
85
|
+
*
|
|
86
|
+
* under `npx vitest run` → …/node_modules/vitest/dist/workers/forks.js
|
|
87
|
+
* under `node foo.eval.mjs`→ …/foo.eval.mjs
|
|
88
|
+
* under `vigiles eval` → the script vigiles spawned
|
|
89
|
+
*
|
|
90
|
+
* So the check is a POSITIVE identification of a known runner, not "am I the entry
|
|
91
|
+
* point". Positive is the conservative direction here: an unrecognised wrapper
|
|
92
|
+
* runs (and may cost money) rather than a legitimate run being blocked by a
|
|
93
|
+
* pattern nobody predicted.
|
|
94
|
+
*
|
|
95
|
+
* ## 🔴 `node --test` HAS NO argv SIGNAL AT ALL, and the line that claimed one
|
|
96
|
+
* was dead from the day it was written
|
|
97
|
+
*
|
|
98
|
+
* `["node --test", ["node_modules/.bin/node--test"]]` used to sit in the table
|
|
99
|
+
* below. It was added by analogy with the npx-installed runners and never
|
|
100
|
+
* measured. There is no such binary: Node's test runner is a FLAG on node
|
|
101
|
+
* itself, so nothing under `node_modules/.bin/` can appear in `argv[1]`, and the
|
|
102
|
+
* entry could not fire under any invocation. That is worse than its absence — it
|
|
103
|
+
* READ as coverage while the paid tier stayed reachable from a legacy harness
|
|
104
|
+
* named `*.test.mjs`, a name Node's own default patterns collect.
|
|
105
|
+
*
|
|
106
|
+
* Measured 2026-08-12 (Node 22.22, a fixture printing its own process facts):
|
|
107
|
+
*
|
|
108
|
+
* node --test foo.test.mjs
|
|
109
|
+
* argv[1] = /abs/foo.test.mjs ← the TEST FILE, no runner in sight
|
|
110
|
+
* execArgv = [] ← the flag is not here either
|
|
111
|
+
* NODE_TEST_CONTEXT = child-v8 ← the only signal
|
|
112
|
+
*
|
|
113
|
+
* node --test --experimental-test-isolation=none foo.test.mjs
|
|
114
|
+
* argv[1] = foo.test.mjs
|
|
115
|
+
* execArgv = ["--test", "--experimental-test-isolation=none"]
|
|
116
|
+
* NODE_TEST_CONTEXT = undefined ← in-process: no child, no var
|
|
117
|
+
*
|
|
118
|
+
* The two modes therefore need two different facts, and both are read here.
|
|
119
|
+
*
|
|
120
|
+
* ⚠️ THIS IS NOT THE `process.env.VITEST` IDIOM REJECTED ABOVE, and the
|
|
121
|
+
* difference is the reason it is allowed. `VITEST` is a USER-VISIBLE CONVENTION:
|
|
122
|
+
* a person can put it in a `.env` or a CI job for unrelated reasons, so a guard
|
|
123
|
+
* reading it fires on correct input. `NODE_TEST_CONTEXT` is set by NODE ITSELF in
|
|
124
|
+
* the child it spawns — the internal protocol between runner and test process,
|
|
125
|
+
* which nobody writes into their own environment. Same for `execArgv`: the flag
|
|
126
|
+
* list node was launched with (`--test` is explicitly refused inside
|
|
127
|
+
* `NODE_OPTIONS` — verified — so it cannot arrive from a stray env either). Do
|
|
128
|
+
* not "fix" this back into an argv fragment: for `node --test` an argv signal
|
|
129
|
+
* does not exist, so the choice is these facts or no detection at all.
|
|
130
|
+
*
|
|
131
|
+
* Pure: facts in, a name or null out. No process access, no filesystem.
|
|
132
|
+
*/
|
|
133
|
+
/**
|
|
134
|
+
* The process facts identifying Node's own test runner, which leaves no trace in
|
|
135
|
+
* `argv[1]`. Passed in rather than read, so the decision stays pure.
|
|
136
|
+
*/
|
|
137
|
+
export interface NodeTestFacts {
|
|
138
|
+
/** `process.execArgv` — carries `--test` when the runner is IN-PROCESS. */
|
|
139
|
+
readonly execArgv?: readonly string[];
|
|
140
|
+
/** `process.env.NODE_TEST_CONTEXT` — node sets it in the test CHILD it spawns. */
|
|
141
|
+
readonly nodeTestContext?: string | undefined;
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* The foreign test runner that started this process, or `null`.
|
|
145
|
+
*
|
|
146
|
+
* `argv1` is `process.argv[1]`; `node` carries the two facts that identify Node's
|
|
147
|
+
* built-in runner (see the header). Both are passed in rather than read, so the
|
|
148
|
+
* decision is a pure function the tests drive with values measured from real runs.
|
|
149
|
+
*/
|
|
150
|
+
export declare function foreignRunner(argv1: string | undefined, node?: NodeTestFacts): string | null;
|
|
151
|
+
/**
|
|
152
|
+
* The refusal message. Separate from the check so a test can assert the WORDS —
|
|
153
|
+
* a guard that stops a run without saying what to do instead is a support ticket.
|
|
154
|
+
*/
|
|
155
|
+
export declare function foreignRunnerRefusal(runner: string, what: string): string;
|
|
156
|
+
/**
|
|
157
|
+
* Refuse to spend model budget when a foreign test runner started this process.
|
|
158
|
+
*
|
|
159
|
+
* 🔴 CALL THIS AT EVERY REAL-MODEL SPAWN. It used to live privately in `eval.ts`
|
|
160
|
+
* and be called from exactly one place — `spawnAgent`, the real `claude` runner —
|
|
161
|
+
* on the reasoning that the composition root funnels every paid path. Measured
|
|
162
|
+
* 2026-08-12: it funnels ONE of four.
|
|
163
|
+
*
|
|
164
|
+
* src/eval.ts spawn(claudeCodeRuntime.agentBinary) guarded
|
|
165
|
+
* src/adapters/codex/eval.ts spawnSync("codex", …) UNGUARDED
|
|
166
|
+
* src/judge.ts spawnSync("claude", …) UNGUARDED
|
|
167
|
+
* src/scan-behavioral.ts spawnSync("claude", …) UNGUARDED
|
|
168
|
+
*
|
|
169
|
+
* `measureTriggerRate(spec, { evalDriver })` calls the injected driver's runner
|
|
170
|
+
* DIRECTLY, so a Codex eval collected by a stray `npx vitest run` spent real
|
|
171
|
+
* money on every push while the guard looked complete. The irony is worth
|
|
172
|
+
* recording: the untested-skill nudge now tells Codex users to pass
|
|
173
|
+
* `{ evalDriver }` — the product recommends the shape that bypassed the guard.
|
|
174
|
+
*
|
|
175
|
+
* Lives here, in a module that imports NOTHING, so any adapter can call it
|
|
176
|
+
* without a cycle. `foreignRunner` stays pure; this is the one impure wrapper,
|
|
177
|
+
* and it is impure precisely so callers cannot forget to pass the facts.
|
|
178
|
+
*
|
|
179
|
+
* ⚠️ CALL IT OUTSIDE ANY `try` THAT SWALLOWS. `judge` and `deriveAttackReal` both
|
|
180
|
+
* wrap their spawn in `try { … } catch { return fallback }`, so a refusal thrown
|
|
181
|
+
* inside would be caught and downgraded to a score of 0 / a canned string — a
|
|
182
|
+
* silent wrong answer instead of a stop. Both call it before the `try`.
|
|
183
|
+
*/
|
|
184
|
+
export declare function refuseUnderForeignRunner(what: string): void;
|
|
185
|
+
//# sourceMappingURL=foreign-runner.d.ts.map
|