switchroom 0.21.9 → 0.21.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -99,9 +99,26 @@ export function codeSpanSafe(s: string): string {
99
99
  * `\` first (so we never double-escape a following escape), then BOTH `(` and
100
100
  * `)` — the whole URL is preserved balanced and micromark decodes it back to
101
101
  * the original href on round-trip. Bot API 10.1 lists `(`/`)` as escapable.
102
+ *
103
+ * WHITESPACE and control characters are PERCENT-ENCODED rather than
104
+ * backslash-escaped. A bare link destination ends at the first ASCII
105
+ * whitespace character, so a space or a newline inside the href does not just
106
+ * truncate the URL — the remainder is re-read as a link title, or (for a
107
+ * newline) the inline link is terminated outright, leaving a structurally
108
+ * broken construct with URL fragments visible as prose. Backslash cannot
109
+ * rescue that: whitespace is not escapable in a bare destination. `%20` /
110
+ * `%0A` are the canonical URL encodings, so the href a client resolves is
111
+ * equivalent to the one the author wrote.
102
112
  */
103
113
  export function escapeLinkHref(href: string): string {
104
- return href.replace(/\\/g, '\\\\').replace(/\(/g, '\\(').replace(/\)/g, '\\)')
114
+ return href
115
+ .replace(/\\/g, '\\\\')
116
+ .replace(/\(/g, '\\(')
117
+ .replace(/\)/g, '\\)')
118
+ .replace(
119
+ /[\x00-\x20\x7f]/g,
120
+ (c) => `%${c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0')}`,
121
+ )
105
122
  }
106
123
 
107
124
  /**
@@ -43,6 +43,7 @@ import {
43
43
  isQuoteRejectionError,
44
44
  sendOptsHaveQuote,
45
45
  } from '../reply-quote.js'
46
+ import { captureSpeechText } from './speech-capture.js'
46
47
 
47
48
  // ── send-orchestration façade imports (#2996 P2) ──
48
49
  // Pure/deterministic helpers are imported; stateful or side-effecting gateway
@@ -1376,6 +1377,12 @@ export async function sendReply(
1376
1377
  // plain-text TTS input); synthesis happens just before the send so a
1377
1378
  // voice-only reply can suppress the text chunk loop on success. Voice is
1378
1379
  // fully best-effort — every failure below falls back to the text reply.
1380
+ // Raw-corpus capture (TTS redesign PR-0, flag-gated, off by default): `text`
1381
+ // here IS Stage A's future input — capture it byte-for-byte BEFORE the
1382
+ // resolve call, ahead of any TTS normalisation, so the redesign's property
1383
+ // tests can validate against real markdown instead of synthetic fixtures
1384
+ // only. See telegram-plugin/gateway/speech-capture.ts.
1385
+ captureSpeechText(text)
1379
1386
  const voiceOutPlan = resolveVoiceOutPlan(access.voice_out, text)
1380
1387
  const configParseMode = access.parseMode ?? 'html'
1381
1388
  const format = (args.format as string | undefined) ?? configParseMode
@@ -0,0 +1,158 @@
1
+ /**
2
+ * Raw pre-normalisation speech-text capture (TTS normalisation redesign,
3
+ * PR-0 — `/tmp/claude-0/tts/out/fable-plan-v2.md` §9).
4
+ *
5
+ * Flag-gated on `SWITCHROOM_SPEECH_CAPTURE=1` (or `=true`), OFF by default:
6
+ * unset (or any other value) is a true no-op — `captureSpeechText` returns
7
+ * immediately without touching the filesystem, so the send path pays zero
8
+ * cost. When enabled, appends exactly one JSON line `{ts, text}` — where
9
+ * `text` is the EXACT string about to be passed to `resolveVoiceOutPlan`,
10
+ * byte-for-byte, no re-encoding — to `$TELEGRAM_STATE_DIR/speech-capture.jsonl`
11
+ * before that call (fable-plan-v2.md §0 C-4: that string IS Stage A's future
12
+ * input). NOT the same string the visible rich render displays — the render
13
+ * path additionally runs `computeEffectiveText`/`addParagraphSpacers` (U+00A0
14
+ * paragraph spacers) downstream of this capture point; this captures the
15
+ * plain pre-spacer prose Stage A will actually consume.
16
+ *
17
+ * Why this exists: `corpus.json` (the existing replay corpus) was captured
18
+ * downstream of the legacy pass-1 normaliser, so it carries zero tables,
19
+ * fences, pipes, or emoji (fable-plan-v2.md §0 C-3) — useless for validating
20
+ * the new Stage-A renderer beyond synthetic fixtures. `telegram/history.db`
21
+ * is also unusable: it stores text as Telegram echoed it back (rendered),
22
+ * not the pre-render markdown. Capture must happen in-process, here.
23
+ *
24
+ * Privacy: the capture file holds the full plaintext body of every outbound
25
+ * reply while the flag is on, so it is created `0o600` (owner-only) — NOT
26
+ * the default-umask `0o644` `appendFileSync` would otherwise produce, which
27
+ * would leave it world-readable in a directory where `access.json` (the chat
28
+ * allowlist) is deliberately `0o600`. Mode only applies at file CREATION
29
+ * (Node ignores `mode` on an append to an existing file), matching the
30
+ * established pattern in this codebase (`shown-ledger.ts`, `outbox.ts`).
31
+ *
32
+ * Write failures are swallowed — capture must never break a send — but never
33
+ * silently: the first write failure (and, rate-limited, subsequent ones)
34
+ * logs one stderr line, because a foreign-uid EACCES on this file (the #4371
35
+ * failure class: some other process touches it, it becomes root-owned, the
36
+ * agent uid EACCESes on every append thereafter) must be observable across a
37
+ * 7-day unattended capture window, not discovered after the fact from an
38
+ * empty corpus.
39
+ *
40
+ * Deliberately does NOT bound or rotate the capture file: the spec (§9 PR-0)
41
+ * describes a plain unbounded append for a time-boxed 7-day fleet-wide
42
+ * capture window (measured fleet-wide: ~2.9 MB over 7 days), after which the
43
+ * file is reviewed, secrets-scrubbed, and checked into the repo as a static
44
+ * fixture — it is not a long-lived production log.
45
+ *
46
+ * The enabled check reads `process.env` on every call (deliberately NOT
47
+ * cached at module scope, unlike `current-turn-map.ts`'s
48
+ * `EMISSION_AUTHORITY_ENABLED` kill-switch convention): the read is a single
49
+ * property lookup, not the per-turn state-store cost that convention exists
50
+ * to avoid, and caching it would make the flag un-togglable within a single
51
+ * test process — which the load-bearing "a capture write failure never
52
+ * breaks the reply" guarantee needs to exercise against the REAL `sendReply`
53
+ * wiring (see `send-reply-golden.test.ts`), not a synthetic call.
54
+ */
55
+
56
+ import { appendFileSync } from 'node:fs'
57
+ import { join } from 'node:path'
58
+
59
+ export const SPEECH_CAPTURE_FILE_NAME = 'speech-capture.jsonl'
60
+
61
+ /** File created owner-read-write only (see the Privacy note above). */
62
+ const SPEECH_CAPTURE_FILE_MODE = 0o600
63
+
64
+ export interface CaptureSpeechTextOptions {
65
+ /** Override the enabled flag (tests only); default reads
66
+ * `SWITCHROOM_SPEECH_CAPTURE` from the environment. */
67
+ enabled?: boolean
68
+ /** Override the destination directory (tests only); default
69
+ * `process.env.TELEGRAM_STATE_DIR`. */
70
+ stateDir?: string
71
+ /** Override the clock (tests only). */
72
+ now?: () => number
73
+ }
74
+
75
+ function envFlagEnabled(): boolean {
76
+ const v = process.env.SWITCHROOM_SPEECH_CAPTURE
77
+ return v === '1' || v === 'true'
78
+ }
79
+
80
+ function isCaptureEnabled(override: boolean | undefined): boolean {
81
+ if (override !== undefined) return override
82
+ return envFlagEnabled()
83
+ }
84
+
85
+ // One-shot "capture is live" log line, latched the first time capture
86
+ // actually fires (not at module import — importing this module must stay a
87
+ // true no-op regardless of whether the flag ends up used) — and a
88
+ // rate-limited write-error log so a foreign-uid EACCES (#4371 class) is
89
+ // observable within the window, not just discoverable from an empty corpus
90
+ // after the fact.
91
+ let loggedEnabledOnce = false
92
+ // `null`, not `0`: a real failure at `Date.now() === 0` (or in a test driven
93
+ // by `vi.setSystemTime(0)`) must still log — `0` is a valid past timestamp,
94
+ // not "never logged", so it cannot double as the sentinel.
95
+ let lastWriteErrorLoggedAt: number | null = null
96
+ const WRITE_ERROR_LOG_INTERVAL_MS = 5 * 60_000
97
+
98
+ function logEnabledOnce(): void {
99
+ if (loggedEnabledOnce) return
100
+ loggedEnabledOnce = true
101
+ try {
102
+ process.stderr.write(
103
+ `telegram gateway: speech-capture: enabled — writing $TELEGRAM_STATE_DIR/${SPEECH_CAPTURE_FILE_NAME}\n`,
104
+ )
105
+ } catch {
106
+ // Logging must never break the send path.
107
+ }
108
+ }
109
+
110
+ function logWriteErrorRateLimited(err: unknown, nowMs: number): void {
111
+ if (lastWriteErrorLoggedAt !== null && nowMs - lastWriteErrorLoggedAt < WRITE_ERROR_LOG_INTERVAL_MS) return
112
+ lastWriteErrorLoggedAt = nowMs
113
+ try {
114
+ const msg = err instanceof Error ? err.message : String(err)
115
+ process.stderr.write(
116
+ `telegram gateway: speech-capture: write failed (further failures rate-limited ` +
117
+ `${Math.round(WRITE_ERROR_LOG_INTERVAL_MS / 60_000)}m) err=${msg}\n`,
118
+ )
119
+ } catch {
120
+ // Logging must never break the send path.
121
+ }
122
+ }
123
+
124
+ /**
125
+ * Append one `{ts, text}` JSON line to the raw-corpus capture file when
126
+ * capture is enabled. `text` must be the EXACT string about to be passed to
127
+ * `resolveVoiceOutPlan` — no trimming, no re-encoding, byte-for-byte
128
+ * preservation of markdown (tables, fences, pipes, emoji, backticks, ...).
129
+ * Off by default; a true no-op when the flag is unset. Never throws.
130
+ */
131
+ export function captureSpeechText(text: string, opts: CaptureSpeechTextOptions = {}): void {
132
+ if (!isCaptureEnabled(opts.enabled)) return
133
+ logEnabledOnce()
134
+ try {
135
+ const stateDir = opts.stateDir ?? process.env.TELEGRAM_STATE_DIR
136
+ if (stateDir == null || stateDir.length === 0) return
137
+ const now = opts.now ?? Date.now
138
+ const line = `${JSON.stringify({ ts: now(), text })}\n`
139
+ appendFileSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME), line, {
140
+ encoding: 'utf8',
141
+ mode: SPEECH_CAPTURE_FILE_MODE,
142
+ })
143
+ } catch (err) {
144
+ // Capture must never break the send path — but never silently either
145
+ // (M2/#4371 class).
146
+ logWriteErrorRateLimited(err, Date.now())
147
+ }
148
+ }
149
+
150
+ /**
151
+ * Test-only: reset the one-shot enable log and the write-error rate limiter,
152
+ * so a test asserting stderr output is independent of import/call order
153
+ * versus every other test sharing this module instance.
154
+ */
155
+ export function __resetSpeechCaptureLogStateForTests(): void {
156
+ loggedEnabledOnce = false
157
+ lastWriteErrorLoggedAt = null
158
+ }
@@ -0,0 +1,354 @@
1
+ // Raw-HTML dialect handling for the Bot API 10.1 rich-markdown render path.
2
+ //
3
+ // ── Why this module exists ────────────────────────────────────────────────
4
+ // Agents habitually emit raw HTML tags in prose (`<b>`, `<i>`, `<a href>`).
5
+ // Before this module those tags reached the wire byte-verbatim, which is two
6
+ // distinct hazards at once:
7
+ //
8
+ // 1. Nothing establishes that Telegram's rich markdown parser accepts them.
9
+ // The wire-verified allowlist (raw `sendRichMessage` probes, 2026-08-13)
10
+ // covers exactly `<u>`, `<sub>`, `<sup>`, `<details>`/`<summary>` and
11
+ // `<aside>`/`<cite>`. `isParseEntitiesError` (`rich-send.ts`) matches
12
+ // `unsupported start tag` / `unclosed start tag`, so the wire CAN 400 on
13
+ // an unknown tag — and that fallback resends the body as PLAIN TEXT,
14
+ // where the reader then sees literal `<b>` markup.
15
+ // 2. mdast `html` nodes degrade to `plain` in `parse.ts`, and `plain` is
16
+ // `escapeMarkdown`'d on render. `escapeMarkdown` escapes `=`, so
17
+ // `<a href="https://example.com">` shipped as `<a href\="…">` — an
18
+ // attribute no parser can read.
19
+ //
20
+ // ── The policy (three buckets, degrade by TYPE, never silently) ───────────
21
+ // • FOLD — a tag with an exact native markdown equivalent is folded
22
+ // into the IR node for that construct, so it renders as the
23
+ // markdown the wire actually understands:
24
+ // <b>/<strong> -> bold **…**
25
+ // <i>/<em> -> italic *…*
26
+ // <s>/<del>/<strike> -> strike ~~…~~
27
+ // <code> -> code span `…`
28
+ // <a href="URL"> -> link [label](URL)
29
+ // <br> -> a line break
30
+ // <pre> -> fenced code block ```…```
31
+ // • PASSTHROUGH — the wire-verified allowlist above is emitted RAW and
32
+ // UNESCAPED (an IR `raw` inline), which is also what stops
33
+ // `escapeMarkdown` from mangling their attributes. Requires a
34
+ // MATCHED close: an unbalanced `<u>` emitted raw is exactly
35
+ // the `unclosed start tag` 400 this module exists to prevent.
36
+ // • DEGRADE — everything else, split by whether the token delimits
37
+ // content. The discriminator is MATCHING, not the tag name:
38
+ // – A MATCHED pair (`<marquee>…</marquee>`, `<div>…</div>`)
39
+ // is markup wrapped around content. The markup is dropped
40
+ // and the content kept; a block-level name additionally
41
+ // leaves a hard line break, so `<li>one</li><li>two</li>`
42
+ // cannot glue into `onetwo`.
43
+ // – A comment, or a void / self-closing tag (`<hr/>`,
44
+ // `<img …/>`), delimits nothing. Markup dropped; a
45
+ // block-level one leaves the same separator.
46
+ // – An UNMATCHED open/close marker is not markup at all — it
47
+ // is PROSE. `<service>/<key>`, `<agent>`, "the `<b>` tag",
48
+ // `run switchroom vault get <key>` are pervasive in this
49
+ // project's own agent output, and DELETING them silently
50
+ // mangles the agent's own reply. Such a token is kept as
51
+ // LITERAL TEXT with `&`/`<`/`>` HTML-entity-escaped (see
52
+ // `escapeHtmlLiteral`). Bot API rich markdown "can contain
53
+ // arbitrary HTML … parsed as described in Rich HTML
54
+ // style", and Rich HTML documents `&lt;`, `&gt;` and
55
+ // `&amp;` among its supported named entities — so the
56
+ // entity form is the DOCUMENTED way to ship a literal
57
+ // angle bracket without tripping `unsupported start tag`.
58
+ //
59
+ // ── The invariant ─────────────────────────────────────────────────────────
60
+ // CONTENT is never lost. Only MARKUP is dropped, and only where dropping it
61
+ // loses no reader-visible text: a matched pair, a comment, a void tag.
62
+ // Anything not demonstrably markup survives to the wire as escaped literal
63
+ // text. (Known residual, NOT introduced by this module and not fixed here: a
64
+ // `<` in prose that mdast hands over as a TEXT node rather than an `html` node
65
+ // — the decoded `&lt;c&gt;`, or `response in <1s` — never reaches this module
66
+ // and still ships unescaped. That is a `plain`-node / `escapeMarkdown`
67
+ // concern.)
68
+ //
69
+ // This is a deterministic code-level guarantee, deliberately not a prompt
70
+ // instruction to the agents that emit the tags.
71
+
72
+ /** How a raw HTML token reads. `other` covers anything the tokenizer matched
73
+ * but could not classify (it degrades like an unknown tag). */
74
+ export type HtmlTagKind = "open" | "close" | "selfclose" | "comment" | "other";
75
+
76
+ export interface HtmlTagInfo {
77
+ kind: HtmlTagKind;
78
+ /** Lowercased tag name (`""` for a comment). */
79
+ name: string;
80
+ /** The raw source bytes of the token, verbatim. */
81
+ raw: string;
82
+ /** Raw attribute text between the tag name and the closing `>`. */
83
+ attrs: string;
84
+ }
85
+
86
+ /** The IR construct a foldable tag maps onto. `pre` is BLOCK-level (a fenced
87
+ * code block); every other target is inline. */
88
+ export type HtmlFoldTarget =
89
+ | "bold"
90
+ | "italic"
91
+ | "strike"
92
+ | "code"
93
+ | "link"
94
+ | "break"
95
+ | "pre";
96
+
97
+ /** Tags with an exact native markdown equivalent on the Bot API 10.1 rich
98
+ * path. Folding them means the wire sees markdown it definitely parses
99
+ * instead of HTML it may reject. */
100
+ export const HTML_FOLD_TAGS: Readonly<Record<string, HtmlFoldTarget>> = {
101
+ b: "bold",
102
+ strong: "bold",
103
+ i: "italic",
104
+ em: "italic",
105
+ s: "strike",
106
+ del: "strike",
107
+ strike: "strike",
108
+ code: "code",
109
+ a: "link",
110
+ br: "break",
111
+ // `<pre>` is the one BLOCK-level fold. Telegram's own Rich HTML reference
112
+ // pairs `<pre><code class="language-…">` with the ```` ```lang ```` fence,
113
+ // so the fenced block is the exact native equivalent. Without it `<pre>`
114
+ // dropped its markup and the inner `<code>` folded to an INLINE span
115
+ // wrapping a newline — a construct Telegram will not parse, leaving the
116
+ // reader literal backticks around broken text.
117
+ pre: "pre",
118
+ };
119
+
120
+ /** HTML elements that carry no content of their own (void elements) plus the
121
+ * XHTML self-closing spelling. Dropping such a token's markup cannot lose
122
+ * text, so it degrades silently rather than surviving as literal prose.
123
+ * `br` is excluded — it FOLDS to a line break. */
124
+ export const HTML_VOID_TAGS: ReadonlySet<string> = new Set([
125
+ "area",
126
+ "base",
127
+ "col",
128
+ "embed",
129
+ "hr",
130
+ "img",
131
+ "input",
132
+ "link",
133
+ "meta",
134
+ "param",
135
+ "source",
136
+ "track",
137
+ "wbr",
138
+ ]);
139
+
140
+ /** Structural / block-level element names. A degrade of one of these leaves a
141
+ * hard line break behind so its neighbours do not glue together
142
+ * (`<ul><li>one</li><li>two</li></ul>` -> `onetwo` was the defect). The
143
+ * passthrough allowlist (`details`, `summary`, `aside`, `cite`) is
144
+ * deliberately absent — those never degrade. */
145
+ export const HTML_BLOCK_LEVEL_TAGS: ReadonlySet<string> = new Set([
146
+ "address",
147
+ "article",
148
+ "blockquote",
149
+ "center",
150
+ "dd",
151
+ "div",
152
+ "dl",
153
+ "dt",
154
+ "fieldset",
155
+ "figcaption",
156
+ "figure",
157
+ "footer",
158
+ "form",
159
+ "h1",
160
+ "h2",
161
+ "h3",
162
+ "h4",
163
+ "h5",
164
+ "h6",
165
+ "header",
166
+ "hr",
167
+ "li",
168
+ "main",
169
+ "nav",
170
+ "ol",
171
+ "p",
172
+ "section",
173
+ "table",
174
+ "tbody",
175
+ "td",
176
+ "tfoot",
177
+ "th",
178
+ "thead",
179
+ "tr",
180
+ "ul",
181
+ ]);
182
+
183
+ /** True when a degraded token should leave a line break behind. */
184
+ export function isBlockLevelTag(name: string): boolean {
185
+ return HTML_BLOCK_LEVEL_TAGS.has(name);
186
+ }
187
+
188
+ /** True for a tag that delimits no content of its own. */
189
+ export function isVoidTag(name: string): boolean {
190
+ return HTML_VOID_TAGS.has(name);
191
+ }
192
+
193
+ /**
194
+ * HTML-entity-escape a run of literal text so its angle brackets reach the
195
+ * reader instead of being parsed as a tag (or 400ing as an unsupported one).
196
+ *
197
+ * Bot API rich markdown "can contain arbitrary HTML … parsed as described in
198
+ * Rich HTML style", and Rich HTML's supported named entities are documented as
199
+ * exactly `&lt; &gt; &amp; &quot; &apos; &nbsp; &hellip; &mdash; &ndash;
200
+ * &lsquo; &rsquo; &ldquo; &rdquo;` (https://core.telegram.org/bots/api). Only
201
+ * the first three are needed here. `&` is escaped FIRST so an already-entity
202
+ * -looking run cannot be double-decoded.
203
+ */
204
+ export function escapeHtmlLiteral(text: string): string {
205
+ return text.replace(/&/g, "&amp;").replace(/</g, "&lt;").replace(/>/g, "&gt;");
206
+ }
207
+
208
+ /** The `class="language-xxx"` hint on a `<pre><code …>` block, or null. */
209
+ const LANGUAGE_CLASS_RE = /(?:^|\s)class\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
210
+
211
+ /** Pull the fenced-block language out of a `<code class="language-python">`
212
+ * tag's attributes, or null. Telegram's reference spells the language hint
213
+ * exactly this way; anything else yields a plain fence. */
214
+ export function languageOf(tag: HtmlTagInfo): string | null {
215
+ const m = LANGUAGE_CLASS_RE.exec(tag.attrs);
216
+ if (m == null) return null;
217
+ const cls = m[1] ?? m[2] ?? m[3] ?? "";
218
+ const lang = cls
219
+ .split(/\s+/)
220
+ .map((c) => /^language-(.+)$/.exec(c)?.[1])
221
+ .find((c): c is string => c != null && c.length > 0);
222
+ // A fence info string cannot contain a backtick (it would close the fence).
223
+ return lang != null && !lang.includes("`") ? lang : null;
224
+ }
225
+
226
+ /** Tags PROVEN on the wire by raw `sendRichMessage` probes (2026-08-13) to
227
+ * render as native typed nodes. These pass through raw and unescaped.
228
+ * Deliberately an allowlist: an unprobed tag is not known-good syntax. */
229
+ export const HTML_PASSTHROUGH_TAGS: ReadonlySet<string> = new Set([
230
+ "u",
231
+ "sub",
232
+ "sup",
233
+ "details",
234
+ "summary",
235
+ "aside",
236
+ "cite",
237
+ ]);
238
+
239
+ /** Matches one HTML comment or one start/end tag. Attribute values containing
240
+ * a raw `>` are not supported (they are invalid unquoted HTML and vanishingly
241
+ * rare in agent prose); such a token simply degrades. */
242
+ export const HTML_TOKEN_RE =
243
+ /<!--[\s\S]*?-->|<\/?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\/?>/g;
244
+
245
+ const TAG_RE = /^<(\/?)([A-Za-z][A-Za-z0-9-]*)((?:\s[^<>]*?)?)(\/?)>$/;
246
+ const HREF_RE = /(?:^|\s)href\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
247
+
248
+ /** Classify a single raw HTML token. */
249
+ export function classifyHtmlTag(raw: string): HtmlTagInfo {
250
+ if (raw.startsWith("<!--")) return { kind: "comment", name: "", raw, attrs: "" };
251
+ const m = TAG_RE.exec(raw);
252
+ if (m == null) return { kind: "other", name: "", raw, attrs: "" };
253
+ const [, slash, name, attrs, selfClose] = m;
254
+ const kind: HtmlTagKind =
255
+ slash === "/" ? "close" : selfClose === "/" ? "selfclose" : "open";
256
+ return { kind, name: name.toLowerCase(), raw, attrs: attrs ?? "" };
257
+ }
258
+
259
+ /** Pull the `href` value out of a tag's attribute text, or null when absent
260
+ * or empty. The ORIGINAL bytes are returned — never rewritten. */
261
+ export function hrefOf(tag: HtmlTagInfo): string | null {
262
+ const m = HREF_RE.exec(tag.attrs);
263
+ if (m == null) return null;
264
+ const value = m[1] ?? m[2] ?? m[3] ?? "";
265
+ return value.length > 0 ? value : null;
266
+ }
267
+
268
+ /** True when this tag is emitted raw and unescaped (wire-verified allowlist). */
269
+ export function isPassthroughTag(tag: HtmlTagInfo): boolean {
270
+ return tag.name.length > 0 && HTML_PASSTHROUGH_TAGS.has(tag.name);
271
+ }
272
+
273
+ /**
274
+ * Decide which HTML tokens in a DOCUMENT are balanced markup and which are
275
+ * bare prose, given every token in document order.
276
+ *
277
+ * Matching cannot be decided inside a single mdast `html` node: micromark
278
+ * splits `<details open>…\n\nbody\n\n</details>` into THREE nodes, so the open
279
+ * and close markers arrive in different streams. A document-level pass is the
280
+ * only place the question is answerable. Feeding it mdast `html` nodes (rather
281
+ * than the raw source) also means tags inside a code fence or a code span are
282
+ * never counted — they are not markup and must not balance anything.
283
+ *
284
+ * Standard HTML-ish matching: opens are pushed on a stack and a close pops to
285
+ * the nearest same-named open, leaving anything above it unmatched. Void and
286
+ * self-closing tags are balanced by definition and are not tracked.
287
+ *
288
+ * Returns the SOURCE OFFSETS of every token that has a partner; a token whose
289
+ * offset is absent is unmatched — prose, per the module policy above.
290
+ */
291
+ export function matchedTagOffsets(
292
+ tokens: ReadonlyArray<{ tag: HtmlTagInfo; start: number }>,
293
+ ): Set<number> {
294
+ const matched = new Set<number>();
295
+ const stack: { name: string; start: number }[] = [];
296
+ for (const { tag, start } of tokens) {
297
+ if (tag.kind === "open" && !isVoidTag(tag.name)) {
298
+ stack.push({ name: tag.name, start });
299
+ continue;
300
+ }
301
+ if (tag.kind !== "close") continue;
302
+ for (let i = stack.length - 1; i >= 0; i--) {
303
+ if (stack[i].name !== tag.name) continue;
304
+ matched.add(stack[i].start);
305
+ matched.add(start);
306
+ stack.length = i;
307
+ break;
308
+ }
309
+ }
310
+ return matched;
311
+ }
312
+
313
+ /** A piece of a raw-HTML string: either an HTML token or a run of text. */
314
+ export type HtmlPiece =
315
+ | { kind: "tag"; tag: HtmlTagInfo; start: number; end: number }
316
+ | { kind: "text"; text: string; start: number; end: number };
317
+
318
+ /**
319
+ * Split a raw string into HTML tokens and the text runs between them.
320
+ * `base` is the absolute source offset the string starts at, so every piece
321
+ * carries usable UTF-16 offsets into the original markdown.
322
+ */
323
+ export function tokenizeHtml(raw: string, base: number): HtmlPiece[] {
324
+ const pieces: HtmlPiece[] = [];
325
+ const re = new RegExp(HTML_TOKEN_RE.source, "g");
326
+ let last = 0;
327
+ let m: RegExpExecArray | null;
328
+ while ((m = re.exec(raw)) != null) {
329
+ if (m.index > last) {
330
+ pieces.push({
331
+ kind: "text",
332
+ text: raw.slice(last, m.index),
333
+ start: base + last,
334
+ end: base + m.index,
335
+ });
336
+ }
337
+ pieces.push({
338
+ kind: "tag",
339
+ tag: classifyHtmlTag(m[0]),
340
+ start: base + m.index,
341
+ end: base + m.index + m[0].length,
342
+ });
343
+ last = m.index + m[0].length;
344
+ }
345
+ if (last < raw.length) {
346
+ pieces.push({
347
+ kind: "text",
348
+ text: raw.slice(last),
349
+ start: base + last,
350
+ end: base + raw.length,
351
+ });
352
+ }
353
+ return pieces;
354
+ }