agent-sanitizer 2.30.0 → 2.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/THREAT-MODEL.md +40 -19
- package/claude-hooks/lib/placeholder-grammar.mjs +162 -84
- package/claude-hooks/lib/reveal.mjs +129 -5
- package/claude-hooks/plugin-hooks.mjs +1 -0
- package/claude-hooks/pretooluse-sanitize.mjs +159 -2
- package/claude-hooks/sanitize-output.mjs +76 -23
- package/package.json +1 -1
- package/src/gates.mjs +5 -3
- package/src/html.mjs +123 -30
- package/src/index.mjs +32 -18
- package/src/output.mjs +73 -17
- package/types/claude-hooks/lib/placeholder-grammar.d.mts +75 -29
- package/types/claude-hooks/lib/reveal.d.mts +46 -0
- package/types/claude-hooks/pretooluse-sanitize.d.mts +35 -0
- package/types/claude-hooks/sanitize-output.d.mts +17 -3
- package/types/gates.d.mts +5 -3
- package/types/html.d.mts +74 -24
- package/types/index.d.mts +19 -10
- package/types/output.d.mts +24 -3
|
@@ -1,11 +1,28 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
*
|
|
2
|
+
* The Layer-2 placeholder keys in `text`, in document order (duplicates kept —
|
|
3
|
+
* callers dedupe when they need to). matchAll clones the global regex, so no
|
|
4
|
+
* lastIndex state leaks between calls.
|
|
5
|
+
* @param {string} text
|
|
6
|
+
* @returns {string[]}
|
|
7
|
+
*/
|
|
8
|
+
export function layer2Keys(text: string): string[];
|
|
9
|
+
/**
|
|
10
|
+
* Every distinct Layer-2 placeholder key anywhere in `value` — the deep-walk
|
|
11
|
+
* twin of {@link layer2Keys}, with the same depth cap (fail OPEN: an advisory
|
|
12
|
+
* miss costs one line, never a mangled input) as {@link containsPlaceholder}.
|
|
13
|
+
* @param {unknown} value
|
|
14
|
+
* @param {number} [depth]
|
|
15
|
+
* @returns {string[]}
|
|
16
|
+
*/
|
|
17
|
+
export function layer2KeysIn(value: unknown, depth?: number): string[];
|
|
18
|
+
/**
|
|
19
|
+
* Depth-capped walk: does any string in `value` carry secret-placeholder-shaped
|
|
20
|
+
* text? The cap fails OPEN (deeper content is unseen) — every caller feeds a
|
|
4
21
|
* context-only advisory, so a miss costs one line, never a mangled input.
|
|
5
22
|
*
|
|
6
23
|
* Kept alongside {@link collectPlaceholders} rather than expressed in terms of
|
|
7
|
-
* it: this one short-circuits on the first hit and ignores the Layer-2
|
|
8
|
-
* which is what the PostToolUse on-disk tripwire wants on every Read.
|
|
24
|
+
* it: this one short-circuits on the first hit and ignores the Layer-2
|
|
25
|
+
* grammar, which is what the PostToolUse on-disk tripwire wants on every Read.
|
|
9
26
|
* @param {unknown} value
|
|
10
27
|
* @param {number} [depth]
|
|
11
28
|
* @returns {boolean}
|
|
@@ -19,8 +36,9 @@ export function containsPlaceholder(value: unknown, depth?: number): boolean;
|
|
|
19
36
|
/**
|
|
20
37
|
* Depth-capped walk collecting every distinct placeholder token in `value`,
|
|
21
38
|
* split by grammar: `secret` for the redaction grammar (PLACEHOLDER_RE),
|
|
22
|
-
* `layer2` for the splice
|
|
23
|
-
* {@link containsPlaceholder} — the
|
|
39
|
+
* `layer2` for the keyed splice placeholders plus the un-keyed unparseable
|
|
40
|
+
* marker. Same cap and fail-OPEN posture as {@link containsPlaceholder} — the
|
|
41
|
+
* consumers are context-only advisories.
|
|
24
42
|
* @param {unknown} value
|
|
25
43
|
* @returns {{ secret: FoundPlaceholder[], layer2: FoundPlaceholder[] }}
|
|
26
44
|
*/
|
|
@@ -30,30 +48,44 @@ export function collectPlaceholders(value: unknown): {
|
|
|
30
48
|
};
|
|
31
49
|
/**
|
|
32
50
|
* Advisory context for a tool call OUTSIDE the rehydrated set (Bash, MCP
|
|
33
|
-
* tools, anything unknown) whose input carries placeholder
|
|
51
|
+
* tools, anything unknown) whose input carries SECRET placeholder text, or
|
|
34
52
|
* null. Rehydration only re-anchors Edit/Write; every other write path — a
|
|
35
53
|
* shell heredoc, `sed -i`, an MCP body field — persists the literal
|
|
36
|
-
* placeholder
|
|
37
|
-
* and the recovery path
|
|
38
|
-
* (`grep` for a placeholder is legitimate), so it is
|
|
39
|
-
* a verdict: a false positive costs a few sentences
|
|
40
|
-
* blocked call or a mangled input.
|
|
54
|
+
* placeholder and destroys the secret it stands for. The advisory names each
|
|
55
|
+
* exact token, the field carrying it, and the recovery path. It cannot tell a
|
|
56
|
+
* write from a read (`grep` for a placeholder is legitimate), so it is
|
|
57
|
+
* deliberately a NOTE, not a verdict: a false positive costs a few sentences
|
|
58
|
+
* of context, never a blocked call or a mangled input.
|
|
41
59
|
*
|
|
42
|
-
* Direct substitution into non-shell tool inputs was evaluated and rejected
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
47
|
-
* map without a named owning file anyway.
|
|
48
|
-
* - Layer-2 markers: the markers are un-keyed, so marker→original is
|
|
49
|
-
* unrecoverable here (the reveal store is addressed by the hash of the full
|
|
50
|
-
* pre-splice text), and blind re-insertion would re-publish hidden
|
|
51
|
-
* untrusted content verbatim.
|
|
60
|
+
* Direct substitution into non-shell tool inputs was evaluated and rejected:
|
|
61
|
+
* substituting the real secret into an MCP body field (a PR body, a comment)
|
|
62
|
+
* would PUBLISH the secret to an external service — exfiltration by
|
|
63
|
+
* construction — and PreToolUse has no placeholder→secret map without a named
|
|
64
|
+
* owning file anyway.
|
|
52
65
|
* @param {string} tool
|
|
53
66
|
* @param {unknown} toolInput
|
|
54
67
|
* @returns {string | null}
|
|
55
68
|
*/
|
|
56
69
|
export function placeholderNotice(tool: string, toolInput: unknown): string | null;
|
|
70
|
+
/**
|
|
71
|
+
* The Layer-2 twin of {@link placeholderNotice}: advisory context for a tool
|
|
72
|
+
* call OUTSIDE the rehydrated set whose input carries Layer-2 splice
|
|
73
|
+
* placeholders, or null. Rehydration restores keyed placeholders to the stored
|
|
74
|
+
* (already-redacted) original only on the Edit/Write path; a shell heredoc,
|
|
75
|
+
* `sed -i`, or an MCP body field persists the placeholder text literally.
|
|
76
|
+
* Deliberately a NOTE, not a verdict, for the same cannot-tell-write-from-read
|
|
77
|
+
* reason as the secret advisory — and it names the span file path(s) where the
|
|
78
|
+
* original bytes live so the model can Read one instead of guessing.
|
|
79
|
+
*
|
|
80
|
+
* Kept separate from {@link placeholderNotice}, rather than folded into it, so
|
|
81
|
+
* that the call site's secret-opt-in gate (with secrets off, `[REDACTED]`-
|
|
82
|
+
* shaped text is ordinary prose) cannot also suppress the Layer-2 advisory:
|
|
83
|
+
* Layer 2 splices regardless of the secret opt-in.
|
|
84
|
+
* @param {string} tool
|
|
85
|
+
* @param {unknown} toolInput
|
|
86
|
+
* @returns {string | null}
|
|
87
|
+
*/
|
|
88
|
+
export function layer2PlaceholderNotice(tool: string, toolInput: unknown): string | null;
|
|
57
89
|
export const PLACEHOLDER_LABEL_CHARS: "A-Za-z0-9 ()._-";
|
|
58
90
|
/**
|
|
59
91
|
* Matches exactly the placeholder text the canonical redactor can emit:
|
|
@@ -63,14 +95,28 @@ export const PLACEHOLDER_LABEL_CHARS: "A-Za-z0-9 ()._-";
|
|
|
63
95
|
*/
|
|
64
96
|
export const PLACEHOLDER_RE: RegExp;
|
|
65
97
|
/**
|
|
66
|
-
* The Layer-2 splice
|
|
67
|
-
*
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
*
|
|
71
|
-
*
|
|
98
|
+
* The keyed Layer-2 splice placeholder grammar — `[hidden HTML removed #<key>]`
|
|
99
|
+
* / `[HTML comment removed #<key>]`, capture group 1 = the 12-lowercase-hex key
|
|
100
|
+
* (the first 12 hex chars of sha256 over the RAW spliced original). Mirrored
|
|
101
|
+
* from the single producer in the engine (`src/html.mjs`,
|
|
102
|
+
* `LAYER2_PLACEHOLDER_RE`) for the same pinned-registry reason as
|
|
103
|
+
* PLACEHOLDER_RE above.
|
|
104
|
+
*
|
|
105
|
+
* Deliberately DISJOINT from the secret-redaction grammar: a Layer-2
|
|
106
|
+
* placeholder never matches PLACEHOLDER_RE (no `[REDACTED` prefix) and a
|
|
107
|
+
* secret placeholder never matches this — so the secret rehydrator and the
|
|
108
|
+
* Layer-2 rehydrator can compose without either touching the other's tokens.
|
|
109
|
+
*/
|
|
110
|
+
export const LAYER2_PLACEHOLDER_RE: RegExp;
|
|
111
|
+
/**
|
|
112
|
+
* The one Layer-2 marker that carries no key, mirrored from src/html.mjs
|
|
113
|
+
* (UNPARSEABLE_PLACEHOLDER) for the bundle-pin reason above. The fail-closed
|
|
114
|
+
* path withholds the WHOLE output, so there is no per-splice original to key:
|
|
115
|
+
* the pre-splice text lives in the reveal sidecar (lib/reveal.mjs) whose exact
|
|
116
|
+
* path the sanitize-time warning named. Not round-trippable — the advisory
|
|
117
|
+
* points at the sidecar instead of a span file.
|
|
72
118
|
*/
|
|
73
|
-
export const
|
|
119
|
+
export const UNPARSEABLE_MARKER: "[HTML unparseable \u2014 withheld]";
|
|
74
120
|
/**
|
|
75
121
|
* One found token: the exact placeholder text and the dotted field path of the
|
|
76
122
|
* FIRST input field carrying it (empty for a bare string input).
|
|
@@ -17,6 +17,45 @@ export function revealDir(): string;
|
|
|
17
17
|
* @returns {string | null}
|
|
18
18
|
*/
|
|
19
19
|
export function persistReveal(content: string): string | null;
|
|
20
|
+
/**
|
|
21
|
+
* The store path for one Layer-2 splice's original bytes, keyed by the
|
|
22
|
+
* placeholder key. Throws on a malformed key (fail loud: every caller extracts
|
|
23
|
+
* the key from LAYER2_PLACEHOLDER_RE, whose capture group can only yield a
|
|
24
|
+
* valid key, so a bad one here is a caller bug, not input).
|
|
25
|
+
* @param {string} key
|
|
26
|
+
* @returns {string}
|
|
27
|
+
*/
|
|
28
|
+
export function spanPath(key: string): string;
|
|
29
|
+
/**
|
|
30
|
+
* Persist one Layer-2 splice's (already-redacted) original under its
|
|
31
|
+
* placeholder key, with the same hardened treatment as {@link persistReveal}:
|
|
32
|
+
* the dir must be a private uid-owned 0700 directory, and the file is created
|
|
33
|
+
* symlink-refusingly (O_EXCL) because the path is content-addressed — an
|
|
34
|
+
* attacker who chose the page bytes can precompute it and pre-plant a symlink.
|
|
35
|
+
* Content-addressed dedupe: an existing entry (same key = same raw original)
|
|
36
|
+
* is left in place and counts as success. Returns true when the span is on
|
|
37
|
+
* disk (written now or already there); false on any failure — non-fatal by
|
|
38
|
+
* contract, exactly like a failed reveal write (the splice already protected
|
|
39
|
+
* the output; a later rehydration of this key fails CLOSED with a deny).
|
|
40
|
+
*
|
|
41
|
+
* The KEY is the caller's, extracted from the placeholder — never recomputed
|
|
42
|
+
* from `content`: the key was minted from the RAW original, and `content` is
|
|
43
|
+
* the redacted original, so a recomputed hash would not match. The key is a
|
|
44
|
+
* NAME, not an integrity check.
|
|
45
|
+
* @param {string} key
|
|
46
|
+
* @param {string} content the splice's original, redacted BEFORE this call
|
|
47
|
+
* @returns {boolean}
|
|
48
|
+
*/
|
|
49
|
+
export function persistSpan(key: string, content: string): boolean;
|
|
50
|
+
/**
|
|
51
|
+
* The stored original for one Layer-2 placeholder key, or null when no span is
|
|
52
|
+
* stored (or the store is unusable). The open refuses symlinks (O_NOFOLLOW):
|
|
53
|
+
* the path is precomputable, so a planted symlink must not let this read pull
|
|
54
|
+
* an arbitrary file's bytes into a rehydrated write.
|
|
55
|
+
* @param {string} key
|
|
56
|
+
* @returns {string | null}
|
|
57
|
+
*/
|
|
58
|
+
export function readSpan(key: string): string | null;
|
|
20
59
|
/**
|
|
21
60
|
* True when this PostToolUse event is a Read of a reveal sidecar file, so its
|
|
22
61
|
* output must be marked untrusted even though Read is otherwise a trusted local
|
|
@@ -29,5 +68,12 @@ export function persistReveal(content: string): string | null;
|
|
|
29
68
|
* @returns {boolean}
|
|
30
69
|
*/
|
|
31
70
|
export function isRevealRead(toolName: string, toolInput: any): boolean;
|
|
71
|
+
/**
|
|
72
|
+
* The model-facing line telling it Layer-2 placeholders round-trip: pushed once
|
|
73
|
+
* per tool output whose splices were persisted, so the model knows to leave the
|
|
74
|
+
* keyed placeholders byte-for-byte intact when copying text back into a file —
|
|
75
|
+
* Edit/Write restore each to the stored original automatically.
|
|
76
|
+
*/
|
|
77
|
+
export const SPAN_ROUNDTRIP_NOTICE: string;
|
|
32
78
|
/** Envelope prepended to a reveal-file Read so its bytes are framed as untrusted. */
|
|
33
79
|
export const REVEAL_READ_ENVELOPE: string;
|
|
@@ -1,3 +1,38 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Layer-2 placeholder rehydration on the write path: an Edit `new_string` or a
|
|
3
|
+
* Write `content` carrying `[hidden HTML removed #<key>]` / `[HTML comment
|
|
4
|
+
* removed #<key>]` placeholders (copied from sanitized tool output) has each
|
|
5
|
+
* one restored to the stored original bytes from the reveal store's span files.
|
|
6
|
+
*
|
|
7
|
+
* SECURITY invariant: the stored span content was REDACTED before persistence
|
|
8
|
+
* (sanitize-output runs strict web-ingress redaction on each splice original
|
|
9
|
+
* before persistSpan), so this rehydration can never write a raw secret — the
|
|
10
|
+
* worst it can restore is `[REDACTED…]` placeholder text standing where the
|
|
11
|
+
* secret was, which the on-disk tripwire then flags on later Reads.
|
|
12
|
+
*
|
|
13
|
+
* Composition with the secret rehydrator: this runs AFTER it (a later terminal
|
|
14
|
+
* layer), and the two grammars are disjoint — a Layer-2 placeholder never
|
|
15
|
+
* matches the `[REDACTED…]` grammar and vice versa — so neither can touch the
|
|
16
|
+
* other's tokens. Running second also means a restored original that contains
|
|
17
|
+
* `[REDACTED…]` text is never re-fed to the secret resolver (which would deny
|
|
18
|
+
* it as a foreign placeholder).
|
|
19
|
+
*
|
|
20
|
+
* `old_string` is deliberately NOT rehydrated: a Layer-2 placeholder there
|
|
21
|
+
* exists in the model's view of PRIOR TOOL OUTPUT, not on disk, so unless the
|
|
22
|
+
* file literally contains the placeholder text, Edit's ordinary no-match
|
|
23
|
+
* failure is the right outcome — no re-anchoring. MultiEdit/NotebookEdit with a
|
|
24
|
+
* Layer-2 placeholder are denied (parity with the secret path: sequential
|
|
25
|
+
* edits / notebook JSON cannot be rehydrated).
|
|
26
|
+
* @param {string} tool
|
|
27
|
+
* @param {any} toolInput
|
|
28
|
+
* @returns {{ updatedInput: any, context: string } | { deny: string } | null}
|
|
29
|
+
*/
|
|
30
|
+
export function rehydrateLayer2(tool: string, toolInput: any): {
|
|
31
|
+
updatedInput: any;
|
|
32
|
+
context: string;
|
|
33
|
+
} | {
|
|
34
|
+
deny: string;
|
|
35
|
+
} | null;
|
|
1
36
|
/**
|
|
2
37
|
* The declared layer chain, layers 2-4. Every entry states the two properties
|
|
3
38
|
* the driver reasons about — whether it ERASES code points another layer reads,
|
|
@@ -50,13 +50,15 @@
|
|
|
50
50
|
* the HTML rewrite (Layer 2) and the exfil-URL scan (Layer 3), the injected
|
|
51
51
|
* secret redactor (Layer 4), and the display-only-SGR carve-out. `reveal` carries
|
|
52
52
|
* the seam's pre-Layer-2 text when the HTML splice removed anything, for the
|
|
53
|
-
* orchestrator to persist
|
|
53
|
+
* orchestrator to persist; `splices` is its per-placeholder twin (each
|
|
54
|
+
* `original` already vetted by the seam's exit redaction, withheld entries
|
|
55
|
+
* dropped there), for the orchestrator's per-key span persistence.
|
|
54
56
|
* @param {string} text
|
|
55
57
|
* @param {string} toolName gates the SGR carve-out and the untrusted-ingress passes
|
|
56
58
|
* @param {{remainingMs: () => number}} [deadline] shared wall-clock budget across
|
|
57
59
|
* all leaves of one hook run; a direct caller gets a fresh full budget
|
|
58
60
|
* @param {SanitizeExtensions} [ext]
|
|
59
|
-
* @returns {Promise<{ cleaned: string, warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
|
|
61
|
+
* @returns {Promise<{ cleaned: string, warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string, splices?: Array<{ placeholder: string, original: string }> }>}
|
|
60
62
|
*/
|
|
61
63
|
export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
62
64
|
remainingMs: () => number;
|
|
@@ -67,6 +69,10 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
|
67
69
|
modified: boolean;
|
|
68
70
|
sgrNote: boolean;
|
|
69
71
|
reveal?: string;
|
|
72
|
+
splices?: Array<{
|
|
73
|
+
placeholder: string;
|
|
74
|
+
original: string;
|
|
75
|
+
}>;
|
|
70
76
|
}>;
|
|
71
77
|
/**
|
|
72
78
|
* Sanitize every string leaf of a tool-output value, preserving its shape.
|
|
@@ -82,6 +88,9 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
|
82
88
|
* `reveals` accumulates each leaf's pre-Layer-2 text (when the HTML splice
|
|
83
89
|
* removed something) for the orchestrator to persist, and `notes` the leaves'
|
|
84
90
|
* NOTE-severity findings — same mutated-accumulator shape as `warnings`.
|
|
91
|
+
* `splices` accumulates each leaf's Layer-2 placeholder→original pairs (the
|
|
92
|
+
* per-key twin of `reveals`, already vetted by the seam) for the orchestrator's
|
|
93
|
+
* span persistence — same mutated-accumulator shape again.
|
|
85
94
|
* @param {any} value
|
|
86
95
|
* @param {string} toolName
|
|
87
96
|
* @param {string[]} warnings
|
|
@@ -91,13 +100,18 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
|
|
|
91
100
|
* @param {SanitizeExtensions} [ext]
|
|
92
101
|
* @param {string[]} [notes] appended last so an existing caller's positional
|
|
93
102
|
* arguments keep their meaning
|
|
103
|
+
* @param {Array<{ placeholder: string, original: string }>} [splices] appended
|
|
104
|
+
* after `notes` for the same positional-compatibility reason
|
|
94
105
|
* @param {string} [path] dotted location of `value` within the tool output,
|
|
95
106
|
* used only to name a key collision's location in its warning
|
|
96
107
|
* @returns {Promise<{ value: any, modified: boolean, sgrNote: boolean }>}
|
|
97
108
|
*/
|
|
98
109
|
export function sanitizeValue(value: any, toolName: string, warnings: string[], reveals?: string[], deadline?: {
|
|
99
110
|
remainingMs: () => number;
|
|
100
|
-
}, ext?: SanitizeExtensions, notes?: string[],
|
|
111
|
+
}, ext?: SanitizeExtensions, notes?: string[], splices?: Array<{
|
|
112
|
+
placeholder: string;
|
|
113
|
+
original: string;
|
|
114
|
+
}>, path?: string): Promise<{
|
|
101
115
|
value: any;
|
|
102
116
|
modified: boolean;
|
|
103
117
|
sgrNote: boolean;
|
package/types/gates.d.mts
CHANGED
|
@@ -31,9 +31,11 @@ export function matchesSecretHint(text: string): boolean;
|
|
|
31
31
|
* Matches any HTML tag-like construct: opening tags, closing tags (`</`),
|
|
32
32
|
* comments and bogus declarations (`<!`), and processing instructions / bogus
|
|
33
33
|
* comments (`<?…?>`, which the HTML tokenizer hides exactly like a comment).
|
|
34
|
-
* The
|
|
35
|
-
*
|
|
36
|
-
*
|
|
34
|
+
* The `<!`/`<?` arms carry a comment-only document into the pipeline at all:
|
|
35
|
+
* without them it would skip both Layer 2's splice of the comment and Layer 3's
|
|
36
|
+
* exfil scan over the comment interior.
|
|
37
|
+
* Gate for Layer 2 (HTML sanitization) and the HTML img/a exfil path in
|
|
38
|
+
* Layer 3.
|
|
37
39
|
*/
|
|
38
40
|
export const HTML_TAG_PRESENT: RegExp;
|
|
39
41
|
/**
|
package/types/html.d.mts
CHANGED
|
@@ -21,18 +21,35 @@ export function isHiddenOpen(htmlValue: string): string | null;
|
|
|
21
21
|
*/
|
|
22
22
|
export function closingTagName(htmlValue: string): string | null;
|
|
23
23
|
/**
|
|
24
|
-
*
|
|
25
|
-
*
|
|
24
|
+
* The keyed, content-addressed placeholder for one Layer-2 splice:
|
|
25
|
+
* `[hidden HTML removed #<key>]` / `[HTML comment removed #<key>]`, where
|
|
26
|
+
* `<key>` is the first 12 lowercase-hex chars of sha256 over the UTF-8
|
|
27
|
+
* encoding of the ORIGINAL spliced text. Content-addressed on purpose:
|
|
28
|
+
* identical spliced content yields the identical placeholder, so a rehydrator
|
|
29
|
+
* can match placeholder → original by key alone, and duplicated content never
|
|
30
|
+
* produces conflicting keys.
|
|
31
|
+
* @param {SpliceKind} kind
|
|
32
|
+
* @param {string} original the exact text the splice removed
|
|
33
|
+
* @returns {string}
|
|
34
|
+
*/
|
|
35
|
+
export function layer2Placeholder(kind: SpliceKind, original: string): string;
|
|
36
|
+
/**
|
|
37
|
+
* Replace each range of `text` with its kind's keyed placeholder, preserving
|
|
38
|
+
* every byte outside the ranges verbatim. Overlapping/nested ranges are merged
|
|
26
39
|
* (defense-in-depth — the scanners emit disjoint ranges).
|
|
40
|
+
*
|
|
41
|
+
* Returns the spliced text plus `pairs`, one per emitted placeholder in output
|
|
42
|
+
* order, each pairing the placeholder with the ORIGINAL bytes it replaced and
|
|
43
|
+
* its start offset in the RETURNED text (UTF-16 code-unit string indices, the
|
|
44
|
+
* same space as `ranges`) — everything a rehydrator needs to undo the splice.
|
|
27
45
|
* @param {string} text
|
|
28
|
-
* @param {
|
|
29
|
-
* @returns {string}
|
|
46
|
+
* @param {SpliceRange[]} ranges
|
|
47
|
+
* @returns {{ text: string, pairs: SplicePair[] }}
|
|
30
48
|
*/
|
|
31
|
-
export function spliceRanges(text: string, ranges:
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
}>): string;
|
|
49
|
+
export function spliceRanges(text: string, ranges: SpliceRange[]): {
|
|
50
|
+
text: string;
|
|
51
|
+
pairs: SplicePair[];
|
|
52
|
+
};
|
|
36
53
|
/**
|
|
37
54
|
* Scan raw HTML for hidden content to strip and preserved tags to report.
|
|
38
55
|
* Returned ranges are offsets into `html`; comments and hidden elements span
|
|
@@ -40,14 +57,10 @@ export function spliceRanges(text: string, ranges: Array<{
|
|
|
40
57
|
* through matching close, and parse5 extends an unclosed element to the end
|
|
41
58
|
* of the fragment — fail-closed for truncated markup).
|
|
42
59
|
* @param {string} html
|
|
43
|
-
* @returns {{ ranges:
|
|
60
|
+
* @returns {{ ranges: SpliceRange[], warned: ReturnType<typeof newWarned> }}
|
|
44
61
|
*/
|
|
45
62
|
export function scanHtmlFragment(html: string): {
|
|
46
|
-
ranges:
|
|
47
|
-
start: number;
|
|
48
|
-
end: number;
|
|
49
|
-
kind: "comment" | "hidden";
|
|
50
|
-
}>;
|
|
63
|
+
ranges: SpliceRange[];
|
|
51
64
|
warned: ReturnType<typeof newWarned>;
|
|
52
65
|
};
|
|
53
66
|
/**
|
|
@@ -59,14 +72,26 @@ export function scanHtmlFragment(html: string): {
|
|
|
59
72
|
export function looksLikeHtmlSource(text: string): boolean;
|
|
60
73
|
/**
|
|
61
74
|
* Layer 2 over web-ingress text: splice out HTML comments and hidden elements
|
|
62
|
-
* (placeholders mark the cuts; all other bytes are preserved verbatim)
|
|
63
|
-
* count preserved scripting/resource tags for the caller's warning. Returns
|
|
64
|
-
* null when there is nothing to strip and nothing to report.
|
|
65
|
-
*
|
|
66
|
-
*
|
|
67
|
-
*
|
|
75
|
+
* (keyed placeholders mark the cuts; all other bytes are preserved verbatim)
|
|
76
|
+
* and count preserved scripting/resource tags for the caller's warning. Returns
|
|
77
|
+
* null when there is nothing to strip and nothing to report.
|
|
78
|
+
*
|
|
79
|
+
* `splices` pairs every emitted placeholder with the original bytes it
|
|
80
|
+
* replaced (see {@link spliceRanges}), so a caller can rehydrate the text —
|
|
81
|
+
* nothing is lost, only hidden behind an identity-carrying placeholder.
|
|
82
|
+
*
|
|
83
|
+
* `unparseable` is set (true) only on the fail-closed path below, where the
|
|
84
|
+
* whole input was withheld behind {@link UNPARSEABLE_PLACEHOLDER} rather than
|
|
85
|
+
* spliced — the caller's warning must describe a whole-output withhold, not a
|
|
86
|
+
* splice. There `splices` is `[]`: the parser blew up before any span could be
|
|
87
|
+
* located, so nothing is recoverable per-splice (the caller's pre-splice
|
|
88
|
+
* `reveal` is the only copy).
|
|
89
|
+
*
|
|
90
|
+
* Idempotent over its own output: a keyed placeholder contains no `<`, so a
|
|
91
|
+
* re-run neither gates on it (HTML_TAG_PRESENT needs a tag) nor reads it as
|
|
92
|
+
* markup — placeholders already in the text pass through byte-identical.
|
|
68
93
|
* @param {string} text
|
|
69
|
-
* @returns {{ text: string, removed: { comments: number, hidden: number }, warned: { tags: Record<string, number>, dataSrc: number }, unparseable?: true } | null}
|
|
94
|
+
* @returns {{ text: string, removed: { comments: number, hidden: number }, warned: { tags: Record<string, number>, dataSrc: number }, splices: SplicePair[], unparseable?: true } | null}
|
|
70
95
|
*/
|
|
71
96
|
export function sanitizeHtml(text: string): {
|
|
72
97
|
text: string;
|
|
@@ -78,6 +103,7 @@ export function sanitizeHtml(text: string): {
|
|
|
78
103
|
tags: Record<string, number>;
|
|
79
104
|
dataSrc: number;
|
|
80
105
|
};
|
|
106
|
+
splices: SplicePair[];
|
|
81
107
|
unparseable?: true;
|
|
82
108
|
} | null;
|
|
83
109
|
/**
|
|
@@ -112,10 +138,34 @@ export function detectExfil(text: string): Array<{
|
|
|
112
138
|
target: string;
|
|
113
139
|
}> | null;
|
|
114
140
|
export const REPORTED_TAGS: Set<string>;
|
|
115
|
-
|
|
116
|
-
|
|
141
|
+
/**
|
|
142
|
+
* The single grammar definition for keyed Layer-2 placeholders — the exact
|
|
143
|
+
* output of {@link layer2Placeholder}, capture group 1 = the key. Global so
|
|
144
|
+
* callers can scan a document for every placeholder; reset `lastIndex` (or
|
|
145
|
+
* use `matchAll`) between uses.
|
|
146
|
+
*/
|
|
147
|
+
export const LAYER2_PLACEHOLDER_RE: RegExp;
|
|
148
|
+
export const HIDDEN_PLACEHOLDER: "[hidden HTML removed";
|
|
149
|
+
export const COMMENT_PLACEHOLDER: "[HTML comment removed";
|
|
117
150
|
export const UNPARSEABLE_PLACEHOLDER: "[HTML unparseable \u2014 withheld]";
|
|
118
151
|
export const DATA_URI_LENGTH_THRESHOLD: 4096;
|
|
152
|
+
export type SpliceKind = "comment" | "hidden";
|
|
153
|
+
export type SpliceRange = {
|
|
154
|
+
start: number;
|
|
155
|
+
end: number;
|
|
156
|
+
kind: SpliceKind;
|
|
157
|
+
};
|
|
158
|
+
/**
|
|
159
|
+
* One splice: the keyed placeholder now in the output text, the ORIGINAL
|
|
160
|
+
* bytes it replaced, and the placeholder's start offset in the RETURNED text.
|
|
161
|
+
* All offsets in this module — unist positions and these — are plain JS
|
|
162
|
+
* string indices, i.e. UTF-16 code units.
|
|
163
|
+
*/
|
|
164
|
+
export type SplicePair = {
|
|
165
|
+
placeholder: string;
|
|
166
|
+
original: string;
|
|
167
|
+
start: number;
|
|
168
|
+
};
|
|
119
169
|
/** @returns {{ tags: Record<string, number>, dataSrc: number }} */
|
|
120
170
|
declare function newWarned(): {
|
|
121
171
|
tags: Record<string, number>;
|
package/types/index.d.mts
CHANGED
|
@@ -21,18 +21,23 @@
|
|
|
21
21
|
*
|
|
22
22
|
* The layer bodies live in `./output.mjs`; this is a facade over them, not a
|
|
23
23
|
* second implementation (see the module doc). It narrows `sanitizeText`'s result
|
|
24
|
-
* to the
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
*
|
|
28
|
-
* Layer
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
24
|
+
* to the fields this entry promises — `modified`/`sgrNote` describe the
|
|
25
|
+
* tool-output pipeline's banner, and `reveal` is produced only by options this
|
|
26
|
+
* facade does not expose. `splices` IS passed through (present only when Layer 2
|
|
27
|
+
* spliced): the placeholder→original pairs a caller needs to rehydrate keyed
|
|
28
|
+
* Layer-2 placeholders — same field, same shape as `sanitizeText`'s, since this
|
|
29
|
+
* facade wraps the same layers (grammar in `./html.mjs`: `layer2Placeholder` /
|
|
30
|
+
* `LAYER2_PLACEHOLDER_RE`). `html` selects Layers 2 AND 3 together here, which
|
|
31
|
+
* is the surface this entry has always had; `exfilScan` exposes Layer 3's
|
|
32
|
+
* non-destructive detection on its own (unconditionally implied by `html`,
|
|
33
|
+
* which it can add to but never switch off) for callers that must keep the
|
|
34
|
+
* visible bytes intact — e.g. a PR diff where the Layer-2 splice would corrupt
|
|
35
|
+
* legitimate markup — matching the separate flags `sanitizeText` takes for the
|
|
36
|
+
* tool-output pipeline, which needs Layer 3's detection without Layer 2's
|
|
37
|
+
* splice.
|
|
33
38
|
* @param {string} text
|
|
34
39
|
* @param {{ html?: boolean, exfilScan?: boolean } | null} [options]
|
|
35
|
-
* @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[] }>}
|
|
40
|
+
* @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], splices?: Array<{ placeholder: string, original: string }> }>}
|
|
36
41
|
*/
|
|
37
42
|
export function sanitize(text: string, options?: {
|
|
38
43
|
html?: boolean;
|
|
@@ -42,6 +47,10 @@ export function sanitize(text: string, options?: {
|
|
|
42
47
|
found: string[];
|
|
43
48
|
warnings: string[];
|
|
44
49
|
notes: string[];
|
|
50
|
+
splices?: Array<{
|
|
51
|
+
placeholder: string;
|
|
52
|
+
original: string;
|
|
53
|
+
}>;
|
|
45
54
|
}>;
|
|
46
55
|
export { applyLayer1, isBenignAnsi, isBenignAnsiKinds, stripAnsiFully, LONE_SURROGATE_RE } from "./layer1.mjs";
|
|
47
56
|
export { stripInvisible, stripInvisibleWithReport, isSgrOnly, STRIP, SGR_RE, CHECKS, CATEGORY, CATEGORY_LABELS, LINGUISTIC_SCRIPTS, VS, BLANK_NON_CF, LONG_RUN_RE, LONG_RUN_THRESHOLD, SCATTERED_THRESHOLD } from "./invisible.mjs";
|
package/types/output.d.mts
CHANGED
|
@@ -43,7 +43,13 @@ export function deleteVerbatimSpans(text: string, spans: string[]): {
|
|
|
43
43
|
* `reveal` is the pre-Layer-2 text, present only when the HTML splice removed
|
|
44
44
|
* bytes, so a caller can persist what was hidden for later inspection (see
|
|
45
45
|
* {@link applyMarkdownPipeline}); the field is omitted otherwise, and also when
|
|
46
|
-
* it could not be vetted (see {@link vetStageValue}).
|
|
46
|
+
* it could not be vetted (see {@link vetStageValue}). `splices` is its
|
|
47
|
+
* per-placeholder twin — Layer 2's placeholder→original pairs, in document
|
|
48
|
+
* order, so a hook can rehydrate individual splices (the keyed-placeholder
|
|
49
|
+
* grammar lives in ./html.mjs: `layer2Placeholder`/`LAYER2_PLACEHOLDER_RE`).
|
|
50
|
+
* Present only when Layer 2 spliced; each `original` is vetted like `reveal`,
|
|
51
|
+
* and one that cannot be vetted is WITHHELD (dropped from the array) under the
|
|
52
|
+
* same doctrine — Layer 4 never saw pre-splice text.
|
|
47
53
|
*
|
|
48
54
|
* Every byte mutation goes through {@link applyMutation} and every Layer-4 call
|
|
49
55
|
* through {@link runRedact}, so a layer cannot re-establish some of the
|
|
@@ -63,7 +69,7 @@ export function deleteVerbatimSpans(text: string, spans: string[]): {
|
|
|
63
69
|
* vocabulary), so they contribute findings only.
|
|
64
70
|
* @param {string} text
|
|
65
71
|
* @param {SanitizeTextOptions} [options]
|
|
66
|
-
* @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
|
|
72
|
+
* @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string, splices?: Array<{ placeholder: string, original: string }> }>}
|
|
67
73
|
*/
|
|
68
74
|
export function sanitizeText(text: string, options?: SanitizeTextOptions): Promise<{
|
|
69
75
|
cleaned: string;
|
|
@@ -73,6 +79,10 @@ export function sanitizeText(text: string, options?: SanitizeTextOptions): Promi
|
|
|
73
79
|
modified: boolean;
|
|
74
80
|
sgrNote: boolean;
|
|
75
81
|
reveal?: string;
|
|
82
|
+
splices?: Array<{
|
|
83
|
+
placeholder: string;
|
|
84
|
+
original: string;
|
|
85
|
+
}>;
|
|
76
86
|
}>;
|
|
77
87
|
/**
|
|
78
88
|
* True only for arrays and PLAIN objects — the two shapes whose contents are
|
|
@@ -105,6 +115,12 @@ export function isWalkableContainer(value: any): boolean;
|
|
|
105
115
|
* the HTML splice removed bytes) so a caller can persist what was hidden — the
|
|
106
116
|
* structured-output analogue of {@link sanitizeText}'s `reveal`. Same
|
|
107
117
|
* mutated-accumulator contract as `warnings`.
|
|
118
|
+
*
|
|
119
|
+
* `splices` accumulates each string leaf's Layer-2 placeholder→original pairs
|
|
120
|
+
* (each `original` already vetted, withheld entries dropped — see
|
|
121
|
+
* {@link sanitizeText}) — the per-placeholder twin of `reveals`, so a hook
|
|
122
|
+
* caller gets them for object-shaped tool output too. Same mutated-accumulator
|
|
123
|
+
* contract as `reveals`.
|
|
108
124
|
* @param {any} value
|
|
109
125
|
* @param {SanitizeTextOptions} options
|
|
110
126
|
* @param {string[]} warnings
|
|
@@ -112,9 +128,14 @@ export function isWalkableContainer(value: any): boolean;
|
|
|
112
128
|
* @param {string[]} [notes] the NOTE-severity counterpart of `warnings`;
|
|
113
129
|
* appended last so an existing positional caller keeps working (it simply
|
|
114
130
|
* discards the notes, which is exactly as loud as before the split)
|
|
131
|
+
* @param {Array<{ placeholder: string, original: string }>} [splices] appended
|
|
132
|
+
* after `notes` for the same positional-compatibility reason
|
|
115
133
|
* @returns {Promise<{ value: any, modified: boolean, sgrNote: boolean }>}
|
|
116
134
|
*/
|
|
117
|
-
export function sanitizeValue(value: any, options: SanitizeTextOptions, warnings: string[], reveals?: string[], notes?: string[]
|
|
135
|
+
export function sanitizeValue(value: any, options: SanitizeTextOptions, warnings: string[], reveals?: string[], notes?: string[], splices?: Array<{
|
|
136
|
+
placeholder: string;
|
|
137
|
+
original: string;
|
|
138
|
+
}>): Promise<{
|
|
118
139
|
value: any;
|
|
119
140
|
modified: boolean;
|
|
120
141
|
sgrNote: boolean;
|