agent-sanitizer 2.30.0 → 2.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,11 +1,28 @@
1
1
  /**
2
- * Depth-capped walk: does any string in `value` carry placeholder-shaped text?
3
- * The cap fails OPEN (deeper content is unseen) — every caller feeds a
2
+ * The Layer-2 placeholder keys in `text`, in document order (duplicates kept —
3
+ * callers dedupe when they need to). matchAll clones the global regex, so no
4
+ * lastIndex state leaks between calls.
5
+ * @param {string} text
6
+ * @returns {string[]}
7
+ */
8
+ export function layer2Keys(text: string): string[];
9
+ /**
10
+ * Every distinct Layer-2 placeholder key anywhere in `value` — the deep-walk
11
+ * twin of {@link layer2Keys}, with the same depth cap (fail OPEN: an advisory
12
+ * miss costs one line, never a mangled input) as {@link containsPlaceholder}.
13
+ * @param {unknown} value
14
+ * @param {number} [depth]
15
+ * @returns {string[]}
16
+ */
17
+ export function layer2KeysIn(value: unknown, depth?: number): string[];
18
+ /**
19
+ * Depth-capped walk: does any string in `value` carry secret-placeholder-shaped
20
+ * text? The cap fails OPEN (deeper content is unseen) — every caller feeds a
4
21
  * context-only advisory, so a miss costs one line, never a mangled input.
5
22
  *
6
23
  * Kept alongside {@link collectPlaceholders} rather than expressed in terms of
7
- * it: this one short-circuits on the first hit and ignores the Layer-2 grammar,
8
- * which is what the PostToolUse on-disk tripwire wants on every Read.
24
+ * it: this one short-circuits on the first hit and ignores the Layer-2
25
+ * grammar, which is what the PostToolUse on-disk tripwire wants on every Read.
9
26
  * @param {unknown} value
10
27
  * @param {number} [depth]
11
28
  * @returns {boolean}
@@ -19,8 +36,9 @@ export function containsPlaceholder(value: unknown, depth?: number): boolean;
19
36
  /**
20
37
  * Depth-capped walk collecting every distinct placeholder token in `value`,
21
38
  * split by grammar: `secret` for the redaction grammar (PLACEHOLDER_RE),
22
- * `layer2` for the splice markers. Same cap and fail-OPEN posture as
23
- * {@link containsPlaceholder} — the consumer is a context-only advisory.
39
+ * `layer2` for the keyed splice placeholders plus the un-keyed unparseable
40
+ * marker. Same cap and fail-OPEN posture as {@link containsPlaceholder} — the
41
+ * consumers are context-only advisories.
24
42
  * @param {unknown} value
25
43
  * @returns {{ secret: FoundPlaceholder[], layer2: FoundPlaceholder[] }}
26
44
  */
@@ -30,30 +48,44 @@ export function collectPlaceholders(value: unknown): {
30
48
  };
31
49
  /**
32
50
  * Advisory context for a tool call OUTSIDE the rehydrated set (Bash, MCP
33
- * tools, anything unknown) whose input carries placeholder-shaped text, or
51
+ * tools, anything unknown) whose input carries SECRET placeholder text, or
34
52
  * null. Rehydration only re-anchors Edit/Write; every other write path — a
35
53
  * shell heredoc, `sed -i`, an MCP body field — persists the literal
36
- * placeholder. The advisory names each exact token, the field carrying it,
37
- * and the recovery path per grammar. It cannot tell a write from a read
38
- * (`grep` for a placeholder is legitimate), so it is deliberately a NOTE, not
39
- * a verdict: a false positive costs a few sentences of context, never a
40
- * blocked call or a mangled input.
54
+ * placeholder and destroys the secret it stands for. The advisory names each
55
+ * exact token, the field carrying it, and the recovery path. It cannot tell a
56
+ * write from a read (`grep` for a placeholder is legitimate), so it is
57
+ * deliberately a NOTE, not a verdict: a false positive costs a few sentences
58
+ * of context, never a blocked call or a mangled input.
41
59
  *
42
- * Direct substitution into non-shell tool inputs was evaluated and rejected —
43
- * this stays an advisory, with no per-tool rehydration allowlist:
44
- * - Secret placeholders: substituting the real secret into an MCP body field
45
- * (a PR body, a comment) would PUBLISH the secret to an external service —
46
- * exfiltration by construction — and PreToolUse has no placeholder→secret
47
- * map without a named owning file anyway.
48
- * - Layer-2 markers: the markers are un-keyed, so marker→original is
49
- * unrecoverable here (the reveal store is addressed by the hash of the full
50
- * pre-splice text), and blind re-insertion would re-publish hidden
51
- * untrusted content verbatim.
60
+ * Direct substitution into non-shell tool inputs was evaluated and rejected:
61
+ * substituting the real secret into an MCP body field (a PR body, a comment)
62
+ * would PUBLISH the secret to an external service — exfiltration by
63
+ * construction — and PreToolUse has no placeholder→secret map without a named
64
+ * owning file anyway.
52
65
  * @param {string} tool
53
66
  * @param {unknown} toolInput
54
67
  * @returns {string | null}
55
68
  */
56
69
  export function placeholderNotice(tool: string, toolInput: unknown): string | null;
70
+ /**
71
+ * The Layer-2 twin of {@link placeholderNotice}: advisory context for a tool
72
+ * call OUTSIDE the rehydrated set whose input carries Layer-2 splice
73
+ * placeholders, or null. Rehydration restores keyed placeholders to the stored
74
+ * (already-redacted) original only on the Edit/Write path; a shell heredoc,
75
+ * `sed -i`, or an MCP body field persists the placeholder text literally.
76
+ * Deliberately a NOTE, not a verdict, for the same cannot-tell-write-from-read
77
+ * reason as the secret advisory — and it names the span file path(s) where the
78
+ * original bytes live so the model can Read one instead of guessing.
79
+ *
80
+ * Kept separate from {@link placeholderNotice}, rather than folded into it, so
81
+ * that the call site's secret-opt-in gate (with secrets off, `[REDACTED]`-
82
+ * shaped text is ordinary prose) cannot also suppress the Layer-2 advisory:
83
+ * Layer 2 splices regardless of the secret opt-in.
84
+ * @param {string} tool
85
+ * @param {unknown} toolInput
86
+ * @returns {string | null}
87
+ */
88
+ export function layer2PlaceholderNotice(tool: string, toolInput: unknown): string | null;
57
89
  export const PLACEHOLDER_LABEL_CHARS: "A-Za-z0-9 ()._-";
58
90
  /**
59
91
  * Matches exactly the placeholder text the canonical redactor can emit:
@@ -63,14 +95,28 @@ export const PLACEHOLDER_LABEL_CHARS: "A-Za-z0-9 ()._-";
63
95
  */
64
96
  export const PLACEHOLDER_RE: RegExp;
65
97
  /**
66
- * The Layer-2 splice markers, mirrored from src/html.mjs
67
- * (COMMENT_PLACEHOLDER / HIDDEN_PLACEHOLDER / UNPARSEABLE_PLACEHOLDER) for the
68
- * bundle-pin reason in the module doc. They are fixed, un-keyed strings: the
69
- * marker itself cannot say WHICH splice it came from — the original text lives
70
- * in the content-addressed reveal sidecar (lib/reveal.mjs) whose exact path
71
- * the sanitize-time warning named.
98
+ * The keyed Layer-2 splice placeholder grammar — `[hidden HTML removed #<key>]`
99
+ * / `[HTML comment removed #<key>]`, capture group 1 = the 12-lowercase-hex key
100
+ * (the first 12 hex chars of sha256 over the RAW spliced original). Mirrored
101
+ * from the single producer in the engine (`src/html.mjs`,
102
+ * `LAYER2_PLACEHOLDER_RE`) for the same pinned-registry reason as
103
+ * PLACEHOLDER_RE above.
104
+ *
105
+ * Deliberately DISJOINT from the secret-redaction grammar: a Layer-2
106
+ * placeholder never matches PLACEHOLDER_RE (no `[REDACTED` prefix) and a
107
+ * secret placeholder never matches this — so the secret rehydrator and the
108
+ * Layer-2 rehydrator can compose without either touching the other's tokens.
109
+ */
110
+ export const LAYER2_PLACEHOLDER_RE: RegExp;
111
+ /**
112
+ * The one Layer-2 marker that carries no key, mirrored from src/html.mjs
113
+ * (UNPARSEABLE_PLACEHOLDER) for the bundle-pin reason above. The fail-closed
114
+ * path withholds the WHOLE output, so there is no per-splice original to key:
115
+ * the pre-splice text lives in the reveal sidecar (lib/reveal.mjs) whose exact
116
+ * path the sanitize-time warning named. Not round-trippable — the advisory
117
+ * points at the sidecar instead of a span file.
72
118
  */
73
- export const LAYER2_PLACEHOLDERS: readonly string[];
119
+ export const UNPARSEABLE_MARKER: "[HTML unparseable \u2014 withheld]";
74
120
  /**
75
121
  * One found token: the exact placeholder text and the dotted field path of the
76
122
  * FIRST input field carrying it (empty for a bare string input).
@@ -17,6 +17,45 @@ export function revealDir(): string;
17
17
  * @returns {string | null}
18
18
  */
19
19
  export function persistReveal(content: string): string | null;
20
+ /**
21
+ * The store path for one Layer-2 splice's original bytes, keyed by the
22
+ * placeholder key. Throws on a malformed key (fail loud: every caller extracts
23
+ * the key from LAYER2_PLACEHOLDER_RE, whose capture group can only yield a
24
+ * valid key, so a bad one here is a caller bug, not input).
25
+ * @param {string} key
26
+ * @returns {string}
27
+ */
28
+ export function spanPath(key: string): string;
29
+ /**
30
+ * Persist one Layer-2 splice's (already-redacted) original under its
31
+ * placeholder key, with the same hardened treatment as {@link persistReveal}:
32
+ * the dir must be a private uid-owned 0700 directory, and the file is created
33
+ * symlink-refusingly (O_EXCL) because the path is content-addressed — an
34
+ * attacker who chose the page bytes can precompute it and pre-plant a symlink.
35
+ * Content-addressed dedupe: an existing entry (same key = same raw original)
36
+ * is left in place and counts as success. Returns true when the span is on
37
+ * disk (written now or already there); false on any failure — non-fatal by
38
+ * contract, exactly like a failed reveal write (the splice already protected
39
+ * the output; a later rehydration of this key fails CLOSED with a deny).
40
+ *
41
+ * The KEY is the caller's, extracted from the placeholder — never recomputed
42
+ * from `content`: the key was minted from the RAW original, and `content` is
43
+ * the redacted original, so a recomputed hash would not match. The key is a
44
+ * NAME, not an integrity check.
45
+ * @param {string} key
46
+ * @param {string} content the splice's original, redacted BEFORE this call
47
+ * @returns {boolean}
48
+ */
49
+ export function persistSpan(key: string, content: string): boolean;
50
+ /**
51
+ * The stored original for one Layer-2 placeholder key, or null when no span is
52
+ * stored (or the store is unusable). The open refuses symlinks (O_NOFOLLOW):
53
+ * the path is precomputable, so a planted symlink must not let this read pull
54
+ * an arbitrary file's bytes into a rehydrated write.
55
+ * @param {string} key
56
+ * @returns {string | null}
57
+ */
58
+ export function readSpan(key: string): string | null;
20
59
  /**
21
60
  * True when this PostToolUse event is a Read of a reveal sidecar file, so its
22
61
  * output must be marked untrusted even though Read is otherwise a trusted local
@@ -29,5 +68,12 @@ export function persistReveal(content: string): string | null;
29
68
  * @returns {boolean}
30
69
  */
31
70
  export function isRevealRead(toolName: string, toolInput: any): boolean;
71
+ /**
72
+ * The model-facing line telling it Layer-2 placeholders round-trip: pushed once
73
+ * per tool output whose splices were persisted, so the model knows to leave the
74
+ * keyed placeholders byte-for-byte intact when copying text back into a file —
75
+ * Edit/Write restore each to the stored original automatically.
76
+ */
77
+ export const SPAN_ROUNDTRIP_NOTICE: string;
32
78
  /** Envelope prepended to a reveal-file Read so its bytes are framed as untrusted. */
33
79
  export const REVEAL_READ_ENVELOPE: string;
@@ -1,3 +1,38 @@
1
+ /**
2
+ * Layer-2 placeholder rehydration on the write path: an Edit `new_string` or a
3
+ * Write `content` carrying `[hidden HTML removed #<key>]` / `[HTML comment
4
+ * removed #<key>]` placeholders (copied from sanitized tool output) has each
5
+ * one restored to the stored original bytes from the reveal store's span files.
6
+ *
7
+ * SECURITY invariant: the stored span content was REDACTED before persistence
8
+ * (sanitize-output runs strict web-ingress redaction on each splice original
9
+ * before persistSpan), so this rehydration can never write a raw secret — the
10
+ * worst it can restore is `[REDACTED…]` placeholder text standing where the
11
+ * secret was, which the on-disk tripwire then flags on later Reads.
12
+ *
13
+ * Composition with the secret rehydrator: this runs AFTER it (a later terminal
14
+ * layer), and the two grammars are disjoint — a Layer-2 placeholder never
15
+ * matches the `[REDACTED…]` grammar and vice versa — so neither can touch the
16
+ * other's tokens. Running second also means a restored original that contains
17
+ * `[REDACTED…]` text is never re-fed to the secret resolver (which would deny
18
+ * it as a foreign placeholder).
19
+ *
20
+ * `old_string` is deliberately NOT rehydrated: a Layer-2 placeholder there
21
+ * exists in the model's view of PRIOR TOOL OUTPUT, not on disk, so unless the
22
+ * file literally contains the placeholder text, Edit's ordinary no-match
23
+ * failure is the right outcome — no re-anchoring. MultiEdit/NotebookEdit with a
24
+ * Layer-2 placeholder are denied (parity with the secret path: sequential
25
+ * edits / notebook JSON cannot be rehydrated).
26
+ * @param {string} tool
27
+ * @param {any} toolInput
28
+ * @returns {{ updatedInput: any, context: string } | { deny: string } | null}
29
+ */
30
+ export function rehydrateLayer2(tool: string, toolInput: any): {
31
+ updatedInput: any;
32
+ context: string;
33
+ } | {
34
+ deny: string;
35
+ } | null;
1
36
  /**
2
37
  * The declared layer chain, layers 2-4. Every entry states the two properties
3
38
  * the driver reasons about — whether it ERASES code points another layer reads,
@@ -50,13 +50,15 @@
50
50
  * the HTML rewrite (Layer 2) and the exfil-URL scan (Layer 3), the injected
51
51
  * secret redactor (Layer 4), and the display-only-SGR carve-out. `reveal` carries
52
52
  * the seam's pre-Layer-2 text when the HTML splice removed anything, for the
53
- * orchestrator to persist.
53
+ * orchestrator to persist; `splices` is its per-placeholder twin (each
54
+ * `original` already vetted by the seam's exit redaction, withheld entries
55
+ * dropped there), for the orchestrator's per-key span persistence.
54
56
  * @param {string} text
55
57
  * @param {string} toolName gates the SGR carve-out and the untrusted-ingress passes
56
58
  * @param {{remainingMs: () => number}} [deadline] shared wall-clock budget across
57
59
  * all leaves of one hook run; a direct caller gets a fresh full budget
58
60
  * @param {SanitizeExtensions} [ext]
59
- * @returns {Promise<{ cleaned: string, warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
61
+ * @returns {Promise<{ cleaned: string, warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string, splices?: Array<{ placeholder: string, original: string }> }>}
60
62
  */
61
63
  export function sanitizeText(text: string, toolName: string, deadline?: {
62
64
  remainingMs: () => number;
@@ -67,6 +69,10 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
67
69
  modified: boolean;
68
70
  sgrNote: boolean;
69
71
  reveal?: string;
72
+ splices?: Array<{
73
+ placeholder: string;
74
+ original: string;
75
+ }>;
70
76
  }>;
71
77
  /**
72
78
  * Sanitize every string leaf of a tool-output value, preserving its shape.
@@ -82,6 +88,9 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
82
88
  * `reveals` accumulates each leaf's pre-Layer-2 text (when the HTML splice
83
89
  * removed something) for the orchestrator to persist, and `notes` the leaves'
84
90
  * NOTE-severity findings — same mutated-accumulator shape as `warnings`.
91
+ * `splices` accumulates each leaf's Layer-2 placeholder→original pairs (the
92
+ * per-key twin of `reveals`, already vetted by the seam) for the orchestrator's
93
+ * span persistence — same mutated-accumulator shape again.
85
94
  * @param {any} value
86
95
  * @param {string} toolName
87
96
  * @param {string[]} warnings
@@ -91,13 +100,18 @@ export function sanitizeText(text: string, toolName: string, deadline?: {
91
100
  * @param {SanitizeExtensions} [ext]
92
101
  * @param {string[]} [notes] appended last so an existing caller's positional
93
102
  * arguments keep their meaning
103
+ * @param {Array<{ placeholder: string, original: string }>} [splices] appended
104
+ * after `notes` for the same positional-compatibility reason
94
105
  * @param {string} [path] dotted location of `value` within the tool output,
95
106
  * used only to name a key collision's location in its warning
96
107
  * @returns {Promise<{ value: any, modified: boolean, sgrNote: boolean }>}
97
108
  */
98
109
  export function sanitizeValue(value: any, toolName: string, warnings: string[], reveals?: string[], deadline?: {
99
110
  remainingMs: () => number;
100
- }, ext?: SanitizeExtensions, notes?: string[], path?: string): Promise<{
111
+ }, ext?: SanitizeExtensions, notes?: string[], splices?: Array<{
112
+ placeholder: string;
113
+ original: string;
114
+ }>, path?: string): Promise<{
101
115
  value: any;
102
116
  modified: boolean;
103
117
  sgrNote: boolean;
package/types/gates.d.mts CHANGED
@@ -31,9 +31,11 @@ export function matchesSecretHint(text: string): boolean;
31
31
  * Matches any HTML tag-like construct: opening tags, closing tags (`</`),
32
32
  * comments and bogus declarations (`<!`), and processing instructions / bogus
33
33
  * comments (`<?…?>`, which the HTML tokenizer hides exactly like a comment).
34
- * The `<?` arm is what lets a PI-only document reach Layer 2's bogus-comment
35
- * splice; without it such a document would skip the pipeline entirely. Gate for
36
- * Layer 2 (HTML sanitization) and the HTML img/a exfil path in Layer 3.
34
+ * The `<!`/`<?` arms carry a comment-only document into the pipeline at all:
35
+ * without them it would skip both Layer 2's splice of the comment and Layer 3's
36
+ * exfil scan over the comment interior.
37
+ * Gate for Layer 2 (HTML sanitization) and the HTML img/a exfil path in
38
+ * Layer 3.
37
39
  */
38
40
  export const HTML_TAG_PRESENT: RegExp;
39
41
  /**
package/types/html.d.mts CHANGED
@@ -21,18 +21,35 @@ export function isHiddenOpen(htmlValue: string): string | null;
21
21
  */
22
22
  export function closingTagName(htmlValue: string): string | null;
23
23
  /**
24
- * Replace each range of `text` with its kind's placeholder, preserving every
25
- * byte outside the ranges verbatim. Overlapping/nested ranges are merged
24
+ * The keyed, content-addressed placeholder for one Layer-2 splice:
25
+ * `[hidden HTML removed #<key>]` / `[HTML comment removed #<key>]`, where
26
+ * `<key>` is the first 12 lowercase-hex chars of sha256 over the UTF-8
27
+ * encoding of the ORIGINAL spliced text. Content-addressed on purpose:
28
+ * identical spliced content yields the identical placeholder, so a rehydrator
29
+ * can match placeholder → original by key alone, and duplicated content never
30
+ * produces conflicting keys.
31
+ * @param {SpliceKind} kind
32
+ * @param {string} original the exact text the splice removed
33
+ * @returns {string}
34
+ */
35
+ export function layer2Placeholder(kind: SpliceKind, original: string): string;
36
+ /**
37
+ * Replace each range of `text` with its kind's keyed placeholder, preserving
38
+ * every byte outside the ranges verbatim. Overlapping/nested ranges are merged
26
39
  * (defense-in-depth — the scanners emit disjoint ranges).
40
+ *
41
+ * Returns the spliced text plus `pairs`, one per emitted placeholder in output
42
+ * order, each pairing the placeholder with the ORIGINAL bytes it replaced and
43
+ * its start offset in the RETURNED text (UTF-16 code-unit string indices, the
44
+ * same space as `ranges`) — everything a rehydrator needs to undo the splice.
27
45
  * @param {string} text
28
- * @param {Array<{start: number, end: number, kind: "comment" | "hidden"}>} ranges
29
- * @returns {string}
46
+ * @param {SpliceRange[]} ranges
47
+ * @returns {{ text: string, pairs: SplicePair[] }}
30
48
  */
31
- export function spliceRanges(text: string, ranges: Array<{
32
- start: number;
33
- end: number;
34
- kind: "comment" | "hidden";
35
- }>): string;
49
+ export function spliceRanges(text: string, ranges: SpliceRange[]): {
50
+ text: string;
51
+ pairs: SplicePair[];
52
+ };
36
53
  /**
37
54
  * Scan raw HTML for hidden content to strip and preserved tags to report.
38
55
  * Returned ranges are offsets into `html`; comments and hidden elements span
@@ -40,14 +57,10 @@ export function spliceRanges(text: string, ranges: Array<{
40
57
  * through matching close, and parse5 extends an unclosed element to the end
41
58
  * of the fragment — fail-closed for truncated markup).
42
59
  * @param {string} html
43
- * @returns {{ ranges: Array<{start: number, end: number, kind: "comment" | "hidden"}>, warned: ReturnType<typeof newWarned> }}
60
+ * @returns {{ ranges: SpliceRange[], warned: ReturnType<typeof newWarned> }}
44
61
  */
45
62
  export function scanHtmlFragment(html: string): {
46
- ranges: Array<{
47
- start: number;
48
- end: number;
49
- kind: "comment" | "hidden";
50
- }>;
63
+ ranges: SpliceRange[];
51
64
  warned: ReturnType<typeof newWarned>;
52
65
  };
53
66
  /**
@@ -59,14 +72,26 @@ export function scanHtmlFragment(html: string): {
59
72
  export function looksLikeHtmlSource(text: string): boolean;
60
73
  /**
61
74
  * Layer 2 over web-ingress text: splice out HTML comments and hidden elements
62
- * (placeholders mark the cuts; all other bytes are preserved verbatim) and
63
- * count preserved scripting/resource tags for the caller's warning. Returns
64
- * null when there is nothing to strip and nothing to report. `unparseable` is
65
- * set (true) only on the fail-closed path below, where the whole input was
66
- * withheld behind {@link UNPARSEABLE_PLACEHOLDER} rather than spliced — the
67
- * caller's warning must describe a whole-output withhold, not a splice.
75
+ * (keyed placeholders mark the cuts; all other bytes are preserved verbatim)
76
+ * and count preserved scripting/resource tags for the caller's warning. Returns
77
+ * null when there is nothing to strip and nothing to report.
78
+ *
79
+ * `splices` pairs every emitted placeholder with the original bytes it
80
+ * replaced (see {@link spliceRanges}), so a caller can rehydrate the text —
81
+ * nothing is lost, only hidden behind an identity-carrying placeholder.
82
+ *
83
+ * `unparseable` is set (true) only on the fail-closed path below, where the
84
+ * whole input was withheld behind {@link UNPARSEABLE_PLACEHOLDER} rather than
85
+ * spliced — the caller's warning must describe a whole-output withhold, not a
86
+ * splice. There `splices` is `[]`: the parser blew up before any span could be
87
+ * located, so nothing is recoverable per-splice (the caller's pre-splice
88
+ * `reveal` is the only copy).
89
+ *
90
+ * Idempotent over its own output: a keyed placeholder contains no `<`, so a
91
+ * re-run neither gates on it (HTML_TAG_PRESENT needs a tag) nor reads it as
92
+ * markup — placeholders already in the text pass through byte-identical.
68
93
  * @param {string} text
69
- * @returns {{ text: string, removed: { comments: number, hidden: number }, warned: { tags: Record<string, number>, dataSrc: number }, unparseable?: true } | null}
94
+ * @returns {{ text: string, removed: { comments: number, hidden: number }, warned: { tags: Record<string, number>, dataSrc: number }, splices: SplicePair[], unparseable?: true } | null}
70
95
  */
71
96
  export function sanitizeHtml(text: string): {
72
97
  text: string;
@@ -78,6 +103,7 @@ export function sanitizeHtml(text: string): {
78
103
  tags: Record<string, number>;
79
104
  dataSrc: number;
80
105
  };
106
+ splices: SplicePair[];
81
107
  unparseable?: true;
82
108
  } | null;
83
109
  /**
@@ -112,10 +138,34 @@ export function detectExfil(text: string): Array<{
112
138
  target: string;
113
139
  }> | null;
114
140
  export const REPORTED_TAGS: Set<string>;
115
- export const COMMENT_PLACEHOLDER: "[HTML comment removed]";
116
- export const HIDDEN_PLACEHOLDER: "[hidden HTML removed]";
141
+ /**
142
+ * The single grammar definition for keyed Layer-2 placeholders — the exact
143
+ * output of {@link layer2Placeholder}, capture group 1 = the key. Global so
144
+ * callers can scan a document for every placeholder; reset `lastIndex` (or
145
+ * use `matchAll`) between uses.
146
+ */
147
+ export const LAYER2_PLACEHOLDER_RE: RegExp;
148
+ export const HIDDEN_PLACEHOLDER: "[hidden HTML removed";
149
+ export const COMMENT_PLACEHOLDER: "[HTML comment removed";
117
150
  export const UNPARSEABLE_PLACEHOLDER: "[HTML unparseable \u2014 withheld]";
118
151
  export const DATA_URI_LENGTH_THRESHOLD: 4096;
152
+ export type SpliceKind = "comment" | "hidden";
153
+ export type SpliceRange = {
154
+ start: number;
155
+ end: number;
156
+ kind: SpliceKind;
157
+ };
158
+ /**
159
+ * One splice: the keyed placeholder now in the output text, the ORIGINAL
160
+ * bytes it replaced, and the placeholder's start offset in the RETURNED text.
161
+ * All offsets in this module — unist positions and these — are plain JS
162
+ * string indices, i.e. UTF-16 code units.
163
+ */
164
+ export type SplicePair = {
165
+ placeholder: string;
166
+ original: string;
167
+ start: number;
168
+ };
119
169
  /** @returns {{ tags: Record<string, number>, dataSrc: number }} */
120
170
  declare function newWarned(): {
121
171
  tags: Record<string, number>;
package/types/index.d.mts CHANGED
@@ -21,18 +21,23 @@
21
21
  *
22
22
  * The layer bodies live in `./output.mjs`; this is a facade over them, not a
23
23
  * second implementation (see the module doc). It narrows `sanitizeText`'s result
24
- * to the four fields this entry promises — `modified`/`sgrNote`
25
- * describe the tool-output pipeline's banner, and `reveal` is produced only by
26
- * options this facade does not expose. `html` selects Layers 2 AND 3 together
27
- * here, which is the surface this entry has always had; `exfilScan` exposes
28
- * Layer 3's non-destructive detection on its own (unconditionally implied by
29
- * `html`, which it can add to but never switch off) for
30
- * callers that must keep the visible bytes intact — e.g. a PR diff where the
31
- * Layer-2 splice would corrupt legitimate markup — matching the separate
32
- * flags `sanitizeText` takes for the tool-output pipeline.
24
+ * to the fields this entry promises — `modified`/`sgrNote` describe the
25
+ * tool-output pipeline's banner, and `reveal` is produced only by options this
26
+ * facade does not expose. `splices` IS passed through (present only when Layer 2
27
+ * spliced): the placeholder→original pairs a caller needs to rehydrate keyed
28
+ * Layer-2 placeholders — same field, same shape as `sanitizeText`'s, since this
29
+ * facade wraps the same layers (grammar in `./html.mjs`: `layer2Placeholder` /
30
+ * `LAYER2_PLACEHOLDER_RE`). `html` selects Layers 2 AND 3 together here, which
31
+ * is the surface this entry has always had; `exfilScan` exposes Layer 3's
32
+ * non-destructive detection on its own (unconditionally implied by `html`,
33
+ * which it can add to but never switch off) for callers that must keep the
34
+ * visible bytes intact — e.g. a PR diff where the Layer-2 splice would corrupt
35
+ * legitimate markup — matching the separate flags `sanitizeText` takes for the
36
+ * tool-output pipeline, which needs Layer 3's detection without Layer 2's
37
+ * splice.
33
38
  * @param {string} text
34
39
  * @param {{ html?: boolean, exfilScan?: boolean } | null} [options]
35
- * @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[] }>}
40
+ * @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], splices?: Array<{ placeholder: string, original: string }> }>}
36
41
  */
37
42
  export function sanitize(text: string, options?: {
38
43
  html?: boolean;
@@ -42,6 +47,10 @@ export function sanitize(text: string, options?: {
42
47
  found: string[];
43
48
  warnings: string[];
44
49
  notes: string[];
50
+ splices?: Array<{
51
+ placeholder: string;
52
+ original: string;
53
+ }>;
45
54
  }>;
46
55
  export { applyLayer1, isBenignAnsi, isBenignAnsiKinds, stripAnsiFully, LONE_SURROGATE_RE } from "./layer1.mjs";
47
56
  export { stripInvisible, stripInvisibleWithReport, isSgrOnly, STRIP, SGR_RE, CHECKS, CATEGORY, CATEGORY_LABELS, LINGUISTIC_SCRIPTS, VS, BLANK_NON_CF, LONG_RUN_RE, LONG_RUN_THRESHOLD, SCATTERED_THRESHOLD } from "./invisible.mjs";
@@ -43,7 +43,13 @@ export function deleteVerbatimSpans(text: string, spans: string[]): {
43
43
  * `reveal` is the pre-Layer-2 text, present only when the HTML splice removed
44
44
  * bytes, so a caller can persist what was hidden for later inspection (see
45
45
  * {@link applyMarkdownPipeline}); the field is omitted otherwise, and also when
46
- * it could not be vetted (see {@link vetStageValue}).
46
+ * it could not be vetted (see {@link vetStageValue}). `splices` is its
47
+ * per-placeholder twin — Layer 2's placeholder→original pairs, in document
48
+ * order, so a hook can rehydrate individual splices (the keyed-placeholder
49
+ * grammar lives in ./html.mjs: `layer2Placeholder`/`LAYER2_PLACEHOLDER_RE`).
50
+ * Present only when Layer 2 spliced; each `original` is vetted like `reveal`,
51
+ * and one that cannot be vetted is WITHHELD (dropped from the array) under the
52
+ * same doctrine — Layer 4 never saw pre-splice text.
47
53
  *
48
54
  * Every byte mutation goes through {@link applyMutation} and every Layer-4 call
49
55
  * through {@link runRedact}, so a layer cannot re-establish some of the
@@ -63,7 +69,7 @@ export function deleteVerbatimSpans(text: string, spans: string[]): {
63
69
  * vocabulary), so they contribute findings only.
64
70
  * @param {string} text
65
71
  * @param {SanitizeTextOptions} [options]
66
- * @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string }>}
72
+ * @returns {Promise<{ cleaned: string, found: string[], warnings: string[], notes: string[], modified: boolean, sgrNote: boolean, reveal?: string, splices?: Array<{ placeholder: string, original: string }> }>}
67
73
  */
68
74
  export function sanitizeText(text: string, options?: SanitizeTextOptions): Promise<{
69
75
  cleaned: string;
@@ -73,6 +79,10 @@ export function sanitizeText(text: string, options?: SanitizeTextOptions): Promi
73
79
  modified: boolean;
74
80
  sgrNote: boolean;
75
81
  reveal?: string;
82
+ splices?: Array<{
83
+ placeholder: string;
84
+ original: string;
85
+ }>;
76
86
  }>;
77
87
  /**
78
88
  * True only for arrays and PLAIN objects — the two shapes whose contents are
@@ -105,6 +115,12 @@ export function isWalkableContainer(value: any): boolean;
105
115
  * the HTML splice removed bytes) so a caller can persist what was hidden — the
106
116
  * structured-output analogue of {@link sanitizeText}'s `reveal`. Same
107
117
  * mutated-accumulator contract as `warnings`.
118
+ *
119
+ * `splices` accumulates each string leaf's Layer-2 placeholder→original pairs
120
+ * (each `original` already vetted, withheld entries dropped — see
121
+ * {@link sanitizeText}) — the per-placeholder twin of `reveals`, so a hook
122
+ * caller gets them for object-shaped tool output too. Same mutated-accumulator
123
+ * contract as `reveals`.
108
124
  * @param {any} value
109
125
  * @param {SanitizeTextOptions} options
110
126
  * @param {string[]} warnings
@@ -112,9 +128,14 @@ export function isWalkableContainer(value: any): boolean;
112
128
  * @param {string[]} [notes] the NOTE-severity counterpart of `warnings`;
113
129
  * appended last so an existing positional caller keeps working (it simply
114
130
  * discards the notes, which is exactly as loud as before the split)
131
+ * @param {Array<{ placeholder: string, original: string }>} [splices] appended
132
+ * after `notes` for the same positional-compatibility reason
115
133
  * @returns {Promise<{ value: any, modified: boolean, sgrNote: boolean }>}
116
134
  */
117
- export function sanitizeValue(value: any, options: SanitizeTextOptions, warnings: string[], reveals?: string[], notes?: string[]): Promise<{
135
+ export function sanitizeValue(value: any, options: SanitizeTextOptions, warnings: string[], reveals?: string[], notes?: string[], splices?: Array<{
136
+ placeholder: string;
137
+ original: string;
138
+ }>): Promise<{
118
139
  value: any;
119
140
  modified: boolean;
120
141
  sgrNote: boolean;