agent-sanitizer 2.25.0 → 2.26.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -6
- package/THREAT-MODEL.md +55 -3
- package/bin/sanitize-cli.mjs +12 -9
- package/claude-hooks/lib/control-plane.mjs +18 -2
- package/claude-hooks/lib/hook-timing.mjs +94 -5
- package/claude-hooks/sanitize-output.mjs +73 -27
- package/claude-hooks/scan-invisible-chars.mjs +31 -86
- package/package.json +1 -1
- package/src/claude-context.mjs +125 -0
- package/src/html.mjs +48 -10
- package/src/index.mjs +6 -5
- package/src/instructions.mjs +36 -6
- package/src/invisible.mjs +85 -5
- package/src/layer1.mjs +15 -0
- package/src/output.mjs +159 -54
- package/src/prompt.mjs +11 -28
- package/src/rehydrate.mjs +12 -7
- package/src/severity.mjs +97 -0
- package/src/view-map.mjs +154 -74
- package/types/claude-context.d.mts +88 -0
- package/types/claude-hooks/lib/hook-timing.d.mts +64 -0
- package/types/claude-hooks/sanitize-output.d.mts +8 -5
- package/types/claude-hooks/scan-invisible-chars.d.mts +13 -31
- package/types/html.d.mts +7 -1
- package/types/index.d.mts +5 -3
- package/types/instructions.d.mts +14 -4
- package/types/invisible.d.mts +66 -0
- package/types/layer1.d.mts +12 -0
- package/types/output.d.mts +29 -14
- package/types/severity.d.mts +83 -0
- package/types/src/claude-context.d.mts +88 -0
- package/types/view-map.d.mts +75 -45
package/src/view-map.mjs
CHANGED
|
@@ -12,9 +12,19 @@
|
|
|
12
12
|
* placeholder (`pairs` from the injected redactor’s map mode)
|
|
13
13
|
*
|
|
14
14
|
* The view is carried by {@link makeFileView}, the ONLY constructor the
|
|
15
|
-
* consumers of this module may use: it
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* consumers of this module may use: it brands the result, and the brand carries
|
|
16
|
+
* the coordinate SPACE the pair offsets live in — `"codePoint"` as the injected
|
|
17
|
+
* redactor emits them, `"utf16"` as every function below indexes by. Conversion
|
|
18
|
+
* is a separate, one-way door ({@link toUtf16View}) that accepts only a
|
|
19
|
+
* `"codePoint"` view and returns a new `"utf16"` one, so converting twice throws
|
|
20
|
+
* instead of shifting every astral-preceded offset a second time. Each consumer
|
|
21
|
+
* asserts the space it needs, so a wrongly-spaced view fails at the boundary
|
|
22
|
+
* rather than resolving onto the wrong bytes.
|
|
23
|
+
*
|
|
24
|
+
* The space is part of the value rather than a convention in a comment because
|
|
25
|
+
* the two spaces are otherwise indistinguishable — bare numbers in a bare
|
|
26
|
+
* object — which is exactly why the original double-conversion bug was silently
|
|
27
|
+
* accepted.
|
|
18
28
|
*/
|
|
19
29
|
|
|
20
30
|
/**
|
|
@@ -28,57 +38,123 @@ const FILE_VIEW = Symbol("agent-sanitizer:file-view");
|
|
|
28
38
|
|
|
29
39
|
/**
|
|
30
40
|
* @typedef {{ placeholder: string, original: string, start: number }} RedactionPair
|
|
31
|
-
* @typedef {
|
|
32
|
-
*
|
|
33
|
-
*
|
|
41
|
+
* @typedef {"codePoint" | "utf16"} OffsetSpace
|
|
42
|
+
* Units a {@link FileView}'s pair offsets are expressed in. `codePoint` is
|
|
43
|
+
* what the redactor's map mode emits (Python indexes strings by code point);
|
|
44
|
+
* `utf16` is what every function in this module indexes by (JS
|
|
45
|
+
* `indexOf`/`slice`/`.length` count UTF-16 code units).
|
|
46
|
+
*/
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* A branded, frozen carrier from {@link makeFileView}, tagged with the space its
|
|
50
|
+
* `pairs` offsets live in. Its own JSDoc block: a `@template` applies to every
|
|
51
|
+
* typedef in the comment it sits in, so sharing one with the two above would
|
|
52
|
+
* make `RedactionPair` and `OffsetSpace` generic too.
|
|
53
|
+
* @template {OffsetSpace} S
|
|
54
|
+
* @typedef {{ readonly space: S, readonly text: string,
|
|
55
|
+
* readonly pairs: readonly RedactionPair[] }} FileView
|
|
34
56
|
*/
|
|
35
57
|
|
|
36
58
|
/**
|
|
37
|
-
*
|
|
59
|
+
* Wrap a redactor's map-mode result in a frozen, branded view tagged with the
|
|
60
|
+
* space its offsets are in.
|
|
38
61
|
*
|
|
39
62
|
* The redactor's own object is never touched. It used to be: the caller did
|
|
40
63
|
* `view.pairs = pairsToUtf16(view.text, view.pairs)`, an in-place mutation of a
|
|
41
64
|
* value returned from an INJECTED seam. A redactor that memoizes its map result
|
|
42
65
|
* (a reasonable thing for a caller to build) hands back the same object on the
|
|
43
|
-
* second identical call, which then
|
|
66
|
+
* second identical call, which then got converted a SECOND time — every
|
|
44
67
|
* placeholder preceded by an astral character shifts again and the same input
|
|
45
|
-
* yields a different verdict.
|
|
46
|
-
* that: the conversion is part of construction, the redactor's value is left
|
|
47
|
-
* alone, and every consumer asserts the brand rather than accepting a
|
|
48
|
-
* hand-assembled `{text, pairs}` whose offsets may or may not be converted.
|
|
68
|
+
* yields a different verdict.
|
|
49
69
|
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
* the
|
|
70
|
+
* Construction no longer converts. It brands and records the space, and
|
|
71
|
+
* {@link toUtf16View} does the conversion behind a check that the input is
|
|
72
|
+
* still in code-point space — so `toUtf16View(alreadyConverted)` throws where
|
|
73
|
+
* `makeFileView(v.text, v.pairs)` on an existing view used to silently convert
|
|
74
|
+
* a second time. That was the one door this carrier left open.
|
|
55
75
|
*
|
|
56
|
-
*
|
|
57
|
-
*
|
|
58
|
-
*
|
|
76
|
+
* Both the array AND each pair object are copied before freezing, so nothing
|
|
77
|
+
* here reaches back into the seam's (possibly memoized) value, and a caller that
|
|
78
|
+
* later mutates its own pairs cannot change what this view resolves.
|
|
79
|
+
* @template {OffsetSpace} S
|
|
59
80
|
* @param {string} text redacted view text
|
|
60
|
-
* @param {RedactionPair[]} pairs redactor pairs, in
|
|
61
|
-
* @
|
|
81
|
+
* @param {readonly RedactionPair[]} pairs redactor pairs, offsets in `space`
|
|
82
|
+
* @param {S} space
|
|
83
|
+
* @returns {FileView<S>}
|
|
62
84
|
*/
|
|
63
|
-
export function makeFileView(text, pairs) {
|
|
85
|
+
export function makeFileView(text, pairs, space) {
|
|
86
|
+
assertPairsOrdered(text, pairs, space);
|
|
64
87
|
return Object.freeze({
|
|
65
88
|
[FILE_VIEW]: true,
|
|
89
|
+
space,
|
|
66
90
|
text,
|
|
67
|
-
pairs: Object.freeze(
|
|
91
|
+
pairs: Object.freeze(pairs.map((pair) => Object.freeze({ ...pair }))),
|
|
68
92
|
});
|
|
69
93
|
}
|
|
70
94
|
|
|
71
95
|
/**
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
*
|
|
75
|
-
*
|
|
96
|
+
* Offsets counted the way `space` counts them: UTF-16 code units, or code
|
|
97
|
+
* points as the Python redactor emits them.
|
|
98
|
+
* @param {string} text
|
|
99
|
+
* @param {OffsetSpace} space
|
|
100
|
+
* @returns {number}
|
|
101
|
+
*/
|
|
102
|
+
function unitLength(text, space) {
|
|
103
|
+
return space === "utf16" ? text.length : Array.from(text).length;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Throw unless `pairs` is in range, sorted by `start` and non-overlapping, with
|
|
108
|
+
* every offset read in `space`.
|
|
109
|
+
*
|
|
110
|
+
* Every consumer here — `mapViewOffset`'s `else break`, `pairDiskSpans`'s
|
|
111
|
+
* sequential walk — assumes exactly this. An out-of-range start indexes past
|
|
112
|
+
* the text and silently yields `undefined`, which then poisons every downstream
|
|
113
|
+
* offset comparison (`undefined < n` is always false); an out-of-order or
|
|
114
|
+
* overlapping pair stops the scan early and mis-maps an offset onto the wrong
|
|
115
|
+
* bytes. Enforced at CONSTRUCTION so no view can exist in that state, rather
|
|
116
|
+
* than at each read.
|
|
117
|
+
* @param {string} text the view text the offsets index into
|
|
118
|
+
* @param {readonly RedactionPair[]} pairs
|
|
119
|
+
* @param {OffsetSpace} space
|
|
120
|
+
* @returns {void}
|
|
121
|
+
*/
|
|
122
|
+
function assertPairsOrdered(text, pairs, space) {
|
|
123
|
+
// The no-secrets rehydration is the common case, and the loop below has
|
|
124
|
+
// nothing to check there. Return before `unitLength`, which for "codePoint"
|
|
125
|
+
// materializes a code-point array over the whole file on every Edit/Write.
|
|
126
|
+
if (pairs.length === 0) return;
|
|
127
|
+
const total = unitLength(text, space);
|
|
128
|
+
// The previous pair's placeholder end. `start < prevEnd` catches an
|
|
129
|
+
// out-of-order start and an overlap in one comparison.
|
|
130
|
+
let prevEnd = 0;
|
|
131
|
+
for (const pair of pairs) {
|
|
132
|
+
if (!Number.isInteger(pair.start) || pair.start < 0 || pair.start > total)
|
|
133
|
+
throw new Error(
|
|
134
|
+
`redaction pair start ${pair.start} is out of range [0, ${total}]`,
|
|
135
|
+
);
|
|
136
|
+
if (pair.start < prevEnd)
|
|
137
|
+
throw new Error(
|
|
138
|
+
`redaction pairs must be sorted and non-overlapping: pair start ${pair.start} precedes previous pair end ${prevEnd}`,
|
|
139
|
+
);
|
|
140
|
+
prevEnd = pair.start + unitLength(pair.placeholder, space);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Throw unless `view` came from {@link makeFileView} AND carries `space`.
|
|
146
|
+
*
|
|
147
|
+
* Two failures, one gate. A hand-rolled `{text, pairs}` has offsets that may or
|
|
148
|
+
* may not have been converted; a real view in the WRONG space has offsets that
|
|
149
|
+
* definitely have not. Either mis-anchors an edit onto the wrong bytes whenever
|
|
150
|
+
* an astral character precedes a placeholder — silently, and only for
|
|
76
151
|
* emoji-bearing files. Fail loudly at the boundary instead.
|
|
77
152
|
* @param {unknown} view
|
|
153
|
+
* @param {OffsetSpace} space the space the calling function indexes by
|
|
78
154
|
* @param {string} fn name of the calling function, for the error
|
|
79
155
|
* @returns {void}
|
|
80
156
|
*/
|
|
81
|
-
function assertFileView(view, fn) {
|
|
157
|
+
function assertFileView(view, space, fn) {
|
|
82
158
|
if (
|
|
83
159
|
view === null ||
|
|
84
160
|
typeof view !== "object" ||
|
|
@@ -88,6 +164,11 @@ function assertFileView(view, fn) {
|
|
|
88
164
|
`${fn} requires a view built by makeFileView(); got a raw object whose ` +
|
|
89
165
|
`pair offsets have not been normalized to UTF-16`,
|
|
90
166
|
);
|
|
167
|
+
const actual = /** @type {FileView<OffsetSpace>} */ (view).space;
|
|
168
|
+
if (actual !== space)
|
|
169
|
+
throw new Error(
|
|
170
|
+
`${fn} requires a view with ${space} pair offsets, got ${actual}`,
|
|
171
|
+
);
|
|
91
172
|
}
|
|
92
173
|
|
|
93
174
|
/**
|
|
@@ -195,53 +276,52 @@ function diskOffset(deletions, cleanedOffset, isEnd) {
|
|
|
195
276
|
* placeholder mis-anchors the edit onto the wrong bytes.
|
|
196
277
|
*
|
|
197
278
|
* Exactly once, though: applying it to its own output shifts every
|
|
198
|
-
* astral-preceded placeholder a second time
|
|
199
|
-
*
|
|
200
|
-
*
|
|
201
|
-
*
|
|
202
|
-
* the
|
|
279
|
+
* astral-preceded placeholder a second time, and bare arrays of numbers give
|
|
280
|
+
* nothing to check that against. Prefer {@link toUtf16View}, which runs this
|
|
281
|
+
* behind a space check so the second application throws instead. This stays
|
|
282
|
+
* exported — it is public API on the `./view-map` subpath, and removing it
|
|
283
|
+
* would be a breaking change the release workflow cannot express (it caps
|
|
284
|
+
* automated bumps at minor) — for callers doing their own offset bookkeeping,
|
|
285
|
+
* who own the once-only discipline themselves.
|
|
203
286
|
* @param {string} text the redacted view text the offsets index into
|
|
204
|
-
* @param {
|
|
205
|
-
* @returns {
|
|
287
|
+
* @param {RedactionPair[]} pairs
|
|
288
|
+
* @returns {RedactionPair[]}
|
|
206
289
|
*/
|
|
207
290
|
export function pairsToUtf16(text, pairs) {
|
|
208
291
|
if (pairs.length === 0) return pairs;
|
|
292
|
+
assertPairsOrdered(text, pairs, "codePoint");
|
|
209
293
|
const codePoints = Array.from(text);
|
|
210
294
|
// prefix[i] = UTF-16 length of the first i code points of `text`.
|
|
211
295
|
const prefix = new Array(codePoints.length + 1);
|
|
212
296
|
prefix[0] = 0;
|
|
213
297
|
for (let i = 0; i < codePoints.length; i++)
|
|
214
298
|
prefix[i + 1] = prefix[i] + codePoints[i].length;
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
);
|
|
242
|
-
prevEnd = pair.start + Array.from(pair.placeholder).length;
|
|
243
|
-
return { ...pair, start: prefix[pair.start] };
|
|
244
|
-
});
|
|
299
|
+
return pairs.map((pair) => ({ ...pair, start: prefix[pair.start] }));
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* The same view with its pair offsets re-expressed in UTF-16 code units — a NEW
|
|
304
|
+
* frozen carrier; `view` is untouched.
|
|
305
|
+
*
|
|
306
|
+
* The one-way door. Only a `"codePoint"` view is accepted, so converting an
|
|
307
|
+
* already-converted view throws rather than shifting every astral-preceded
|
|
308
|
+
* offset a second time — which either mis-anchors the edit or rejects it as
|
|
309
|
+
* cutting a placeholder, both from an input that was fine the first time it was
|
|
310
|
+
* seen. Offset range/sort/overlap validity is not this door's job — every
|
|
311
|
+
* carrier is checked at construction (see {@link assertPairsOrdered}), in
|
|
312
|
+
* whichever space it declares.
|
|
313
|
+
* @param {FileView<"codePoint">} view
|
|
314
|
+
* @returns {FileView<"utf16">}
|
|
315
|
+
*/
|
|
316
|
+
export function toUtf16View(view) {
|
|
317
|
+
assertFileView(view, "codePoint", "toUtf16View");
|
|
318
|
+
// Copied because `view.pairs` is readonly and `pairsToUtf16` keeps the
|
|
319
|
+
// published mutable signature; makeFileView copies again on the way in.
|
|
320
|
+
return makeFileView(
|
|
321
|
+
view.text,
|
|
322
|
+
pairsToUtf16(view.text, [...view.pairs]),
|
|
323
|
+
"utf16",
|
|
324
|
+
);
|
|
245
325
|
}
|
|
246
326
|
|
|
247
327
|
/**
|
|
@@ -275,7 +355,7 @@ function mapViewOffset(pairs, offset) {
|
|
|
275
355
|
* and a mis-attributed run would mis-anchor the edit.
|
|
276
356
|
* @param {string} content disk file content
|
|
277
357
|
* @param {string} cleaned Layer-1 view of `content`
|
|
278
|
-
* @param {FileView} view
|
|
358
|
+
* @param {FileView<"utf16">} view
|
|
279
359
|
* @param {{start: number, deleted: string}[]} deletions
|
|
280
360
|
* @param {number} viewStart
|
|
281
361
|
* @param {number} viewEnd
|
|
@@ -288,7 +368,7 @@ export function resolveSpan(
|
|
|
288
368
|
viewStart,
|
|
289
369
|
viewEnd,
|
|
290
370
|
) {
|
|
291
|
-
assertFileView(view, "resolveSpan");
|
|
371
|
+
assertFileView(view, "utf16", "resolveSpan");
|
|
292
372
|
const cleanedStart = mapViewOffset(view.pairs, viewStart);
|
|
293
373
|
const cleanedEnd = mapViewOffset(view.pairs, viewEnd);
|
|
294
374
|
if (cleanedStart === null || cleanedEnd === null) return null;
|
|
@@ -378,15 +458,15 @@ export function spliceOrdered(text, matches, replacementFor) {
|
|
|
378
458
|
* was never part of the secret); interior runs are included. Callers use these
|
|
379
459
|
* to detect an edit whose on-disk footprint intrudes into bytes the model was
|
|
380
460
|
* never shown.
|
|
381
|
-
* @param {FileView} view
|
|
461
|
+
* @param {FileView<"utf16">} view
|
|
382
462
|
* @param {{start: number, deleted: string}[]} deletions
|
|
383
463
|
* @returns {{start: number, end: number}[]}
|
|
384
464
|
*/
|
|
385
465
|
export function pairDiskSpans(view, deletions) {
|
|
386
|
-
assertFileView(view, "pairDiskSpans");
|
|
466
|
+
assertFileView(view, "utf16", "pairDiskSpans");
|
|
387
467
|
return view.pairs.map((pair) => {
|
|
388
468
|
// pair.start is a placeholder boundary, and makeFileView rejected any pair
|
|
389
|
-
// set
|
|
469
|
+
// set out of order or overlapping (see assertPairsOrdered), so it is never
|
|
390
470
|
// strictly interior to another placeholder: mapViewOffset always resolves.
|
|
391
471
|
// The throw is kept anyway, and is NOT dead weight — it is the difference
|
|
392
472
|
// between crashing and corrupting. `null + pair.original.length` is a
|
|
@@ -395,9 +475,9 @@ export function pairDiskSpans(view, deletions) {
|
|
|
395
475
|
// i.e. an edit footprint pointing at the wrong bytes.
|
|
396
476
|
const cleanedStart = mapViewOffset(view.pairs, pair.start);
|
|
397
477
|
/* c8 ignore start -- unreachable through makeFileView, which rejects the
|
|
398
|
-
overlapping pair set that is the only way to produce null here (see
|
|
399
|
-
constructor test in test/view-map.test.mjs);
|
|
400
|
-
against a future regression in that
|
|
478
|
+
overlapping pair set that is the only way to produce null here (see
|
|
479
|
+
assertPairsOrdered and the constructor test in test/view-map.test.mjs);
|
|
480
|
+
kept as a fail-loud guard against a future regression in that check. `ignore next N` does
|
|
401
481
|
NOT suppress the branch here — only the statement — so the range form is
|
|
402
482
|
required to keep the src branch floor at 100%. */
|
|
403
483
|
if (cleanedStart === null)
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The one directory no instruction-file walk ever descends into. Its own
|
|
3
|
+
* function so the name is spelled once, and so the two predicates that need it
|
|
4
|
+
* (a plain glob walk, and {@link excludeFromContextScan}) cannot disagree.
|
|
5
|
+
* @param {string} entry a bare entry name or a path relative to the scan root
|
|
6
|
+
* @returns {boolean}
|
|
7
|
+
*/
|
|
8
|
+
export function excludeNodeModules(entry: string): boolean;
|
|
9
|
+
/**
|
|
10
|
+
* Entries a context scan must not descend into or return: `node_modules`, and
|
|
11
|
+
* every child of a `.claude` directory that is not whitelisted context.
|
|
12
|
+
*
|
|
13
|
+
* The globs alone would already refuse to MATCH those files, but a glob walker
|
|
14
|
+
* calls this on directories as it walks and prunes the ones it rejects — which
|
|
15
|
+
* is where the cost actually is. Without the prune, a `.claude/worktrees/`
|
|
16
|
+
* holding a few repo checkouts is walked in full on every session start (and,
|
|
17
|
+
* because a doubled-star segment does cross into a dot directory when the
|
|
18
|
+
* pattern names one, a `.claude` NESTED inside a worktree was matched and
|
|
19
|
+
* scanned as if it were this session's context).
|
|
20
|
+
*
|
|
21
|
+
* A walker calls this with both bare names and root-relative paths, so it must
|
|
22
|
+
* answer for either; a bare name carries no `.claude` context and is judged only
|
|
23
|
+
* against `node_modules`.
|
|
24
|
+
* @param {string} entry a bare entry name or a path relative to the scan root
|
|
25
|
+
* @returns {boolean}
|
|
26
|
+
*/
|
|
27
|
+
export function excludeFromContextScan(entry: string): boolean;
|
|
28
|
+
/**
|
|
29
|
+
* WHICH files an agent loads as model context, as data: the glob set and the
|
|
30
|
+
* walk-pruning predicate that together define "everything Claude Code reads as
|
|
31
|
+
* instructions, and nothing else".
|
|
32
|
+
*
|
|
33
|
+
* This is the SINGLE SOURCE for that scope. It used to live inside
|
|
34
|
+
* `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
|
|
35
|
+
* knew the answer and nobody else did: `src/instructions.mjs` takes
|
|
36
|
+
* caller-supplied globs by design (no agent's convention is baked into the
|
|
37
|
+
* engine), so the CLI, the Python port and every downstream fork spelled their
|
|
38
|
+
* own approximation of this list — and an approximation that drifts either
|
|
39
|
+
* scans bulk data that can never reach the model (the 30-second session start
|
|
40
|
+
* this whitelist exists to fix) or MISSES a context directory entirely, which
|
|
41
|
+
* is a silent hole in the one scan standing between a poisoned instruction file
|
|
42
|
+
* and a session that loads it.
|
|
43
|
+
*
|
|
44
|
+
* It is a standalone, dependency-free DATA module (like ./cf-charset.mjs) for
|
|
45
|
+
* two reasons: `src/instructions.mjs` re-exports it as the library's public
|
|
46
|
+
* door, and the hook imports it RELATIVELY — deliberately not through the
|
|
47
|
+
* `agent-sanitizer` specifier the plugin bundle pins to a published engine.
|
|
48
|
+
* This scope is hook POLICY, not engine behavior: it must ship and move with the
|
|
49
|
+
* hook that walks it, or a plugin built against an older pin would prune the
|
|
50
|
+
* wrong directories while believing it had scanned everything.
|
|
51
|
+
*/
|
|
52
|
+
/**
|
|
53
|
+
* The `.claude/` subdirectories whose markdown Claude Code loads as model
|
|
54
|
+
* context. This is a WHITELIST, and that is the point: `.claude/` is also where
|
|
55
|
+
* tooling parks bulk data that is never loaded as context — `worktrees/`
|
|
56
|
+
* (entire checked-out copies of the repo), plus caches, transcripts and
|
|
57
|
+
* snapshots — and globbing `.claude/**` swept all of it in. On a repo with a few
|
|
58
|
+
* populated worktrees that is thousands of files READ at every session start:
|
|
59
|
+
* one report put it at 30 seconds of blocked startup, paid for scanning files
|
|
60
|
+
* that cannot reach the model.
|
|
61
|
+
*
|
|
62
|
+
* A whitelist, not a `worktrees` denylist, because the failure modes are not
|
|
63
|
+
* symmetric: an unlisted context directory costs a scan nobody asked for anyway
|
|
64
|
+
* (the PostToolUse sanitizer still cleans those bytes when a tool reads them),
|
|
65
|
+
* while an unlisted BULK directory silently costs every future session its
|
|
66
|
+
* startup. Add an entry here when Claude Code starts loading a new `.claude/`
|
|
67
|
+
* subdirectory as context.
|
|
68
|
+
*/
|
|
69
|
+
export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
70
|
+
/**
|
|
71
|
+
* Every glob whose matches Claude Code loads as model context: the
|
|
72
|
+
* per-directory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and
|
|
73
|
+
* the whitelisted `.claude/` markdown. Claude Code loads these on entry to their
|
|
74
|
+
* containing directory — a load path that bypasses the PostToolUse sanitizer —
|
|
75
|
+
* so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model
|
|
76
|
+
* uncleaned unless something scans it here.
|
|
77
|
+
*
|
|
78
|
+
* `**` does not descend into dot directories, so NESTED `.claude/` trees need
|
|
79
|
+
* their own doubled-star-prefixed patterns: without them a directory-scoped
|
|
80
|
+
* skill at `packages/foo/.claude/skills/x/SKILL.md` — model context by the same
|
|
81
|
+
* load path — is never matched. That same rule is why the root `.claude` needs
|
|
82
|
+
* no separate entry: a leading doubled star matches zero segments, so the
|
|
83
|
+
* nested patterns cover the root tree too.
|
|
84
|
+
*
|
|
85
|
+
* Pair with {@link excludeFromContextScan}: the patterns alone already refuse to
|
|
86
|
+
* MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
|
|
87
|
+
*/
|
|
88
|
+
export const CLAUDE_INSTRUCTION_GLOBS: readonly string[];
|
|
@@ -1,3 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Milliseconds as the seconds string every notice below prints.
|
|
3
|
+
*
|
|
4
|
+
* Rounds tenths half-UP from an exact integer count of hundredths, rather than
|
|
5
|
+
* `(ms / 1000).toFixed(1)`: the shell port of this module
|
|
6
|
+
* (plugin/scripts/lib/hook-timing.sh) has to produce the byte-identical string
|
|
7
|
+
* with integer arithmetic, and `toFixed` rounds the underlying double — so 1150
|
|
8
|
+
* would print "1.1" here (1.15 is below its decimal value as a double) and "1.2"
|
|
9
|
+
* there. `ms / 100` lands exactly on a half only when `ms` ends in 50, and every
|
|
10
|
+
* such quotient is dyadic, so this rounding is exact for every input.
|
|
11
|
+
* @param {number} ms
|
|
12
|
+
* @returns {string}
|
|
13
|
+
*/
|
|
14
|
+
export function formatSeconds(ms: number): string;
|
|
1
15
|
/**
|
|
2
16
|
* Run `work`, charging its whole duration to provisioning so no timer running
|
|
3
17
|
* across it counts that time. Charged in a `finally`, so a provisioning step
|
|
@@ -37,6 +51,43 @@ export function startHookTimer(now?: () => number): () => number;
|
|
|
37
51
|
* @returns {string | null}
|
|
38
52
|
*/
|
|
39
53
|
export function slowHookNotice(hookName: string, elapsedMs: number, thresholdMs?: number): string | null;
|
|
54
|
+
/**
|
|
55
|
+
* The line for a ONE-TIME provisioning step that overran
|
|
56
|
+
* {@link SLOW_PROVISION_THRESHOLD_MS}, or null when it did not.
|
|
57
|
+
*
|
|
58
|
+
* Deliberately NOT {@link slowHookNotice} with a bigger threshold: that message
|
|
59
|
+
* says "every affected call pays it", which is false here and would send the
|
|
60
|
+
* reader hunting a per-call cost that does not exist. What is actionable about a
|
|
61
|
+
* slow install is the installer (uv resolves in a fraction of pip's time) and
|
|
62
|
+
* the fact that a repeat means the idempotence check is broken — so this asks
|
|
63
|
+
* for a report only on the repeat, which is the version of this that is a bug.
|
|
64
|
+
*
|
|
65
|
+
* The one caller is the shell provisioner, whose port of this module
|
|
66
|
+
* (plugin/scripts/lib/hook-timing.sh) must emit this exact string; that port and
|
|
67
|
+
* this definition are pinned to each other by a contract test rather than left
|
|
68
|
+
* as two independently-worded copies.
|
|
69
|
+
* @param {string} stepName
|
|
70
|
+
* @param {number} elapsedMs
|
|
71
|
+
* @param {number} [thresholdMs]
|
|
72
|
+
* @returns {string | null}
|
|
73
|
+
*/
|
|
74
|
+
export function slowProvisionNotice(stepName: string, elapsedMs: number, thresholdMs?: number): string | null;
|
|
75
|
+
/**
|
|
76
|
+
* Write the slow-hook notice to stderr and return it, or return null when the
|
|
77
|
+
* run was within budget (writing nothing, so the quiet path stays quiet).
|
|
78
|
+
*
|
|
79
|
+
* The one place the notice reaches stderr: every reporter below needs the
|
|
80
|
+
* transcript copy, and a hook whose run ENDED IN AN ERROR has nothing but this —
|
|
81
|
+
* its verdict is the fail-closed one its `onError` composed, and diluting that
|
|
82
|
+
* message with a performance aside would bury the fault. A judge that spent
|
|
83
|
+
* thirty seconds and then threw is exactly the case the timing exists to name,
|
|
84
|
+
* so the error path measures and reports; it just reports on the human channel.
|
|
85
|
+
* @param {string} hookName
|
|
86
|
+
* @param {number} elapsedMs
|
|
87
|
+
* @param {(chunk: string) => void} [writeErr] injectable stderr sink, for tests
|
|
88
|
+
* @returns {string | null}
|
|
89
|
+
*/
|
|
90
|
+
export function writeSlowHookNotice(hookName: string, elapsedMs: number, writeErr?: (chunk: string) => void): string | null;
|
|
40
91
|
/**
|
|
41
92
|
* `verdict` with the slow-hook notice folded into its `additional_context`, or
|
|
42
93
|
* the verdict untouched when the run was within budget. Also writes the notice
|
|
@@ -104,3 +155,16 @@ export function reportSlowHook(hookName: string, elapsedMs: number, hookEventNam
|
|
|
104
155
|
* means something is actually wrong, not that the machine is busy.
|
|
105
156
|
*/
|
|
106
157
|
export const SLOW_HOOK_THRESHOLD_MS: 1000;
|
|
158
|
+
/**
|
|
159
|
+
* Wall-clock a ONE-TIME provisioning step may spend before it is reported.
|
|
160
|
+
*
|
|
161
|
+
* Two orders of magnitude above {@link SLOW_HOOK_THRESHOLD_MS}, because it
|
|
162
|
+
* measures something categorically different: a dependency install that a
|
|
163
|
+
* session pays once, not a cost every tool call repeats. A cold `uv` install of
|
|
164
|
+
* the redactor engine is seconds and a cold `pip` one can be tens of them, so a
|
|
165
|
+
* budget anywhere near a second would report every first session — the alert
|
|
166
|
+
* fatigue this whole module exists to avoid. Past a minute, something is
|
|
167
|
+
* actually wrong (a serial pip resolve, a wedged mirror, or an idempotence bug
|
|
168
|
+
* re-provisioning every session), which is worth saying out loud.
|
|
169
|
+
*/
|
|
170
|
+
export const SLOW_PROVISION_THRESHOLD_MS: 60000;
|
|
@@ -56,13 +56,14 @@
|
|
|
56
56
|
* @param {{remainingMs: () => number}} [deadline] shared wall-clock budget across
|
|
57
57
|
* all leaves of one hook run; a direct caller gets a fresh full budget
|
|
58
58
|
* @param {SanitizeExtensions} [ext]
|
|
59
|
-
* @returns {Promise<{ cleaned: string, warnings: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
|
|
59
|
+
* @returns {Promise<{ cleaned: string, warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
|
|
60
60
|
*/
|
|
61
61
|
export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
62
62
|
remainingMs: () => number;
|
|
63
63
|
}, ext?: SanitizeExtensions): Promise<{
|
|
64
64
|
cleaned: string;
|
|
65
65
|
warnings: string[];
|
|
66
|
+
notes: string[];
|
|
66
67
|
modified: boolean;
|
|
67
68
|
sgrNote: boolean;
|
|
68
69
|
reveal?: string;
|
|
@@ -77,10 +78,10 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
|
77
78
|
* too (a connector can hide a secret in a field name); non-string leaves
|
|
78
79
|
* (booleans, numbers, null) pass through untouched, and `warnings` accumulates
|
|
79
80
|
* across leaves.
|
|
80
|
-
* `sgrNote` is the OR across leaves: true when some leaf
|
|
81
|
+
* `sgrNote` is the OR across leaves: true when some leaf came back note-only.
|
|
81
82
|
* `reveals` accumulates each leaf's pre-Layer-2 text (when the HTML splice
|
|
82
|
-
* removed something) for the orchestrator to persist
|
|
83
|
-
* shape as `warnings`.
|
|
83
|
+
* removed something) for the orchestrator to persist, and `notes` the leaves'
|
|
84
|
+
* NOTE-severity findings — same mutated-accumulator shape as `warnings`.
|
|
84
85
|
* @param {any} value
|
|
85
86
|
* @param {string} toolName
|
|
86
87
|
* @param {string[]} warnings
|
|
@@ -88,11 +89,13 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
|
88
89
|
* @param {{remainingMs: () => number}} [deadline] shared wall-clock budget across
|
|
89
90
|
* every leaf of this value (created once by the top-level caller)
|
|
90
91
|
* @param {SanitizeExtensions} [ext]
|
|
92
|
+
* @param {string[]} [notes] appended last so an existing caller's positional
|
|
93
|
+
* arguments keep their meaning
|
|
91
94
|
* @returns {Promise<{ value: any, modified: boolean, sgrNote: boolean }>}
|
|
92
95
|
*/
|
|
93
96
|
export function sanitizeValue(value: any, toolName: string, warnings: string[], reveals?: string[], deadline?: {
|
|
94
97
|
remainingMs: () => number;
|
|
95
|
-
}, ext?: SanitizeExtensions): Promise<{
|
|
98
|
+
}, ext?: SanitizeExtensions, notes?: string[]): Promise<{
|
|
96
99
|
value: any;
|
|
97
100
|
modified: boolean;
|
|
98
101
|
sgrNote: boolean;
|
|
@@ -71,24 +71,6 @@ export function cliMain(opts?: {
|
|
|
71
71
|
trace?: import("./lib/trace.mjs").TraceFn;
|
|
72
72
|
scan?: () => ReturnType<typeof scanProject>;
|
|
73
73
|
}): Promise<void>;
|
|
74
|
-
/**
|
|
75
|
-
* The `.claude/` subdirectories whose markdown Claude Code loads as model
|
|
76
|
-
* context. This is a WHITELIST, and that is the point: `.claude/` is also where
|
|
77
|
-
* tooling parks bulk data that is never loaded as context — `worktrees/`
|
|
78
|
-
* (entire checked-out copies of the repo), plus caches, transcripts and
|
|
79
|
-
* snapshots — and globbing `.claude/**` swept all of it in. On a repo with a few
|
|
80
|
-
* populated worktrees that is thousands of files READ at every session start:
|
|
81
|
-
* one report put it at 30 seconds of blocked startup, paid for scanning files
|
|
82
|
-
* that cannot reach the model.
|
|
83
|
-
*
|
|
84
|
-
* A whitelist, not a `worktrees` denylist, because the failure modes are not
|
|
85
|
-
* symmetric: an unlisted context directory costs a scan this hook was never
|
|
86
|
-
* asked for anyway (the PostToolUse sanitizer still cleans those bytes when a
|
|
87
|
-
* tool reads them), while an unlisted BULK directory silently costs every future
|
|
88
|
-
* session its startup. Add an entry here when Claude Code starts loading a new
|
|
89
|
-
* `.claude/` subdirectory as context.
|
|
90
|
-
*/
|
|
91
|
-
export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
92
74
|
/**
|
|
93
75
|
* @param {string} filePath
|
|
94
76
|
* @returns {Array<{ line: number, charCount: number, method: string, decoded: string }>}
|
|
@@ -99,6 +81,8 @@ export function scanFile(filePath: string): Array<{
|
|
|
99
81
|
method: string;
|
|
100
82
|
decoded: string;
|
|
101
83
|
}>;
|
|
84
|
+
import { CLAUDE_CONTEXT_SUBDIRS } from "../src/claude-context.mjs";
|
|
85
|
+
import { CLAUDE_INSTRUCTION_GLOBS } from "../src/claude-context.mjs";
|
|
102
86
|
/**
|
|
103
87
|
* @param {string} run
|
|
104
88
|
* @returns {{ method: string, decoded: string }}
|
|
@@ -109,19 +93,17 @@ export function decodeRun(run: string): {
|
|
|
109
93
|
};
|
|
110
94
|
/**
|
|
111
95
|
* Every file under `dir` that Claude Code loads as model context: the
|
|
112
|
-
*
|
|
113
|
-
* whitelisted `.claude/` markdown
|
|
114
|
-
*
|
|
115
|
-
*
|
|
116
|
-
*
|
|
117
|
-
* here. Skips node_modules.
|
|
96
|
+
* per-directory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and
|
|
97
|
+
* the whitelisted `.claude/` markdown. Claude Code loads these on entry to their
|
|
98
|
+
* containing directory — a load path that bypasses the PostToolUse sanitizer —
|
|
99
|
+
* so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model
|
|
100
|
+
* uncleaned unless it is scanned here.
|
|
118
101
|
*
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
122
|
-
*
|
|
123
|
-
*
|
|
124
|
-
* patterns cover the root tree too.
|
|
102
|
+
* The scope itself — which globs, and which directories the walk must prune —
|
|
103
|
+
* is the library's {@link CLAUDE_INSTRUCTION_GLOBS} /
|
|
104
|
+
* {@link excludeFromContextScan}, so this hook and every other consumer read one
|
|
105
|
+
* list (see src/claude-context.mjs for why it is imported relatively rather than
|
|
106
|
+
* through the `agent-sanitizer` specifier the plugin bundle pins).
|
|
125
107
|
* @param {string} dir
|
|
126
108
|
* @returns {string[]}
|
|
127
109
|
*/
|
|
@@ -147,4 +129,4 @@ export function formatReport(allFindings: Array<{
|
|
|
147
129
|
decoded: string;
|
|
148
130
|
}>;
|
|
149
131
|
}>): string;
|
|
150
|
-
export { ALERT_FILE, ALERT_ACK_FILE };
|
|
132
|
+
export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, ALERT_FILE, ALERT_ACK_FILE };
|
package/types/html.d.mts
CHANGED
|
@@ -93,11 +93,17 @@ export function urlHost(url: string): string;
|
|
|
93
93
|
* and HTML attributes (src/href/background/srcset/ping, form action/formaction,
|
|
94
94
|
* meta-refresh). Detection only — the text is never modified; the caller
|
|
95
95
|
* surfaces the threats as a warning.
|
|
96
|
+
*
|
|
97
|
+
* `autoFetched` marks a threat that needs no deliberate act to fire — a
|
|
98
|
+
* rendered image, a stylesheet, a form target, a meta refresh — as opposed to a
|
|
99
|
+
* link somebody has to follow. Both are reported; the caller uses it to decide
|
|
100
|
+
* how loudly (see the exfil tier in ./output.mjs).
|
|
96
101
|
* @param {string} text
|
|
97
|
-
* @returns {Array<{ isImage: boolean, reason: string, target: string }> | null}
|
|
102
|
+
* @returns {Array<{ isImage: boolean, autoFetched: boolean, reason: string, target: string }> | null}
|
|
98
103
|
*/
|
|
99
104
|
export function detectExfil(text: string): Array<{
|
|
100
105
|
isImage: boolean;
|
|
106
|
+
autoFetched: boolean;
|
|
101
107
|
reason: string;
|
|
102
108
|
target: string;
|
|
103
109
|
}> | null;
|