switchroom 0.21.9 → 0.21.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +1 -1
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/dist/gateway/gateway.js +772 -266
- package/telegram-plugin/format.ts +18 -1
- package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
- package/telegram-plugin/gateway/speech-capture.ts +158 -0
- package/telegram-plugin/render/html-fold.ts +354 -0
- package/telegram-plugin/render/parse.ts +570 -29
- package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
- package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
- package/telegram-plugin/tests/render/parse.test.ts +9 -5
- package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
- package/telegram-plugin/tests/speech-capture.test.ts +296 -0
- package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
- package/telegram-plugin/tts-normalize.ts +47 -9
- package/telegram-plugin/voice-normalize-text.ts +48 -9
|
@@ -99,9 +99,26 @@ export function codeSpanSafe(s: string): string {
|
|
|
99
99
|
* `\` first (so we never double-escape a following escape), then BOTH `(` and
|
|
100
100
|
* `)` — the whole URL is preserved balanced and micromark decodes it back to
|
|
101
101
|
* the original href on round-trip. Bot API 10.1 lists `(`/`)` as escapable.
|
|
102
|
+
*
|
|
103
|
+
* WHITESPACE and control characters are PERCENT-ENCODED rather than
|
|
104
|
+
* backslash-escaped. A bare link destination ends at the first ASCII
|
|
105
|
+
* whitespace character, so a space or a newline inside the href does not just
|
|
106
|
+
* truncate the URL — the remainder is re-read as a link title, or (for a
|
|
107
|
+
* newline) the inline link is terminated outright, leaving a structurally
|
|
108
|
+
* broken construct with URL fragments visible as prose. Backslash cannot
|
|
109
|
+
* rescue that: whitespace is not escapable in a bare destination. `%20` /
|
|
110
|
+
* `%0A` are the canonical URL encodings, so the href a client resolves is
|
|
111
|
+
* equivalent to the one the author wrote.
|
|
102
112
|
*/
|
|
103
113
|
export function escapeLinkHref(href: string): string {
|
|
104
|
-
return href
|
|
114
|
+
return href
|
|
115
|
+
.replace(/\\/g, '\\\\')
|
|
116
|
+
.replace(/\(/g, '\\(')
|
|
117
|
+
.replace(/\)/g, '\\)')
|
|
118
|
+
.replace(
|
|
119
|
+
/[\x00-\x20\x7f]/g,
|
|
120
|
+
(c) => `%${c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0')}`,
|
|
121
|
+
)
|
|
105
122
|
}
|
|
106
123
|
|
|
107
124
|
/**
|
|
@@ -43,6 +43,7 @@ import {
|
|
|
43
43
|
isQuoteRejectionError,
|
|
44
44
|
sendOptsHaveQuote,
|
|
45
45
|
} from '../reply-quote.js'
|
|
46
|
+
import { captureSpeechText } from './speech-capture.js'
|
|
46
47
|
|
|
47
48
|
// ── send-orchestration façade imports (#2996 P2) ──
|
|
48
49
|
// Pure/deterministic helpers are imported; stateful or side-effecting gateway
|
|
@@ -1376,6 +1377,12 @@ export async function sendReply(
|
|
|
1376
1377
|
// plain-text TTS input); synthesis happens just before the send so a
|
|
1377
1378
|
// voice-only reply can suppress the text chunk loop on success. Voice is
|
|
1378
1379
|
// fully best-effort — every failure below falls back to the text reply.
|
|
1380
|
+
// Raw-corpus capture (TTS redesign PR-0, flag-gated, off by default): `text`
|
|
1381
|
+
// here IS Stage A's future input — capture it byte-for-byte BEFORE the
|
|
1382
|
+
// resolve call, ahead of any TTS normalisation, so the redesign's property
|
|
1383
|
+
// tests can validate against real markdown instead of synthetic fixtures
|
|
1384
|
+
// only. See telegram-plugin/gateway/speech-capture.ts.
|
|
1385
|
+
captureSpeechText(text)
|
|
1379
1386
|
const voiceOutPlan = resolveVoiceOutPlan(access.voice_out, text)
|
|
1380
1387
|
const configParseMode = access.parseMode ?? 'html'
|
|
1381
1388
|
const format = (args.format as string | undefined) ?? configParseMode
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Raw pre-normalisation speech-text capture (TTS normalisation redesign,
|
|
3
|
+
* PR-0 — `/tmp/claude-0/tts/out/fable-plan-v2.md` §9).
|
|
4
|
+
*
|
|
5
|
+
* Flag-gated on `SWITCHROOM_SPEECH_CAPTURE=1` (or `=true`), OFF by default:
|
|
6
|
+
* unset (or any other value) is a true no-op — `captureSpeechText` returns
|
|
7
|
+
* immediately without touching the filesystem, so the send path pays zero
|
|
8
|
+
* cost. When enabled, appends exactly one JSON line `{ts, text}` — where
|
|
9
|
+
* `text` is the EXACT string about to be passed to `resolveVoiceOutPlan`,
|
|
10
|
+
* byte-for-byte, no re-encoding — to `$TELEGRAM_STATE_DIR/speech-capture.jsonl`
|
|
11
|
+
* before that call (fable-plan-v2.md §0 C-4: that string IS Stage A's future
|
|
12
|
+
* input). NOT the same string the visible rich render displays — the render
|
|
13
|
+
* path additionally runs `computeEffectiveText`/`addParagraphSpacers` (U+00A0
|
|
14
|
+
* paragraph spacers) downstream of this capture point; this captures the
|
|
15
|
+
* plain pre-spacer prose Stage A will actually consume.
|
|
16
|
+
*
|
|
17
|
+
* Why this exists: `corpus.json` (the existing replay corpus) was captured
|
|
18
|
+
* downstream of the legacy pass-1 normaliser, so it carries zero tables,
|
|
19
|
+
* fences, pipes, or emoji (fable-plan-v2.md §0 C-3) — useless for validating
|
|
20
|
+
* the new Stage-A renderer beyond synthetic fixtures. `telegram/history.db`
|
|
21
|
+
* is also unusable: it stores text as Telegram echoed it back (rendered),
|
|
22
|
+
* not the pre-render markdown. Capture must happen in-process, here.
|
|
23
|
+
*
|
|
24
|
+
* Privacy: the capture file holds the full plaintext body of every outbound
|
|
25
|
+
* reply while the flag is on, so it is created `0o600` (owner-only) — NOT
|
|
26
|
+
* the default-umask `0o644` `appendFileSync` would otherwise produce, which
|
|
27
|
+
* would leave it world-readable in a directory where `access.json` (the chat
|
|
28
|
+
* allowlist) is deliberately `0o600`. Mode only applies at file CREATION
|
|
29
|
+
* (Node ignores `mode` on an append to an existing file), matching the
|
|
30
|
+
* established pattern in this codebase (`shown-ledger.ts`, `outbox.ts`).
|
|
31
|
+
*
|
|
32
|
+
* Write failures are swallowed — capture must never break a send — but never
|
|
33
|
+
* silently: the first write failure (and, rate-limited, subsequent ones)
|
|
34
|
+
* logs one stderr line, because a foreign-uid EACCES on this file (the #4371
|
|
35
|
+
* failure class: some other process touches it, it becomes root-owned, the
|
|
36
|
+
* agent uid EACCESes on every append thereafter) must be observable across a
|
|
37
|
+
* 7-day unattended capture window, not discovered after the fact from an
|
|
38
|
+
* empty corpus.
|
|
39
|
+
*
|
|
40
|
+
* Deliberately does NOT bound or rotate the capture file: the spec (§9 PR-0)
|
|
41
|
+
* describes a plain unbounded append for a time-boxed 7-day fleet-wide
|
|
42
|
+
* capture window (measured fleet-wide: ~2.9 MB over 7 days), after which the
|
|
43
|
+
* file is reviewed, secrets-scrubbed, and checked into the repo as a static
|
|
44
|
+
* fixture — it is not a long-lived production log.
|
|
45
|
+
*
|
|
46
|
+
* The enabled check reads `process.env` on every call (deliberately NOT
|
|
47
|
+
* cached at module scope, unlike `current-turn-map.ts`'s
|
|
48
|
+
* `EMISSION_AUTHORITY_ENABLED` kill-switch convention): the read is a single
|
|
49
|
+
* property lookup, not the per-turn state-store cost that convention exists
|
|
50
|
+
* to avoid, and caching it would make the flag un-togglable within a single
|
|
51
|
+
* test process — which the load-bearing "a capture write failure never
|
|
52
|
+
* breaks the reply" guarantee needs to exercise against the REAL `sendReply`
|
|
53
|
+
* wiring (see `send-reply-golden.test.ts`), not a synthetic call.
|
|
54
|
+
*/
|
|
55
|
+
|
|
56
|
+
import { appendFileSync } from 'node:fs'
|
|
57
|
+
import { join } from 'node:path'
|
|
58
|
+
|
|
59
|
+
export const SPEECH_CAPTURE_FILE_NAME = 'speech-capture.jsonl'
|
|
60
|
+
|
|
61
|
+
/** File created owner-read-write only (see the Privacy note above). */
|
|
62
|
+
const SPEECH_CAPTURE_FILE_MODE = 0o600
|
|
63
|
+
|
|
64
|
+
export interface CaptureSpeechTextOptions {
|
|
65
|
+
/** Override the enabled flag (tests only); default reads
|
|
66
|
+
* `SWITCHROOM_SPEECH_CAPTURE` from the environment. */
|
|
67
|
+
enabled?: boolean
|
|
68
|
+
/** Override the destination directory (tests only); default
|
|
69
|
+
* `process.env.TELEGRAM_STATE_DIR`. */
|
|
70
|
+
stateDir?: string
|
|
71
|
+
/** Override the clock (tests only). */
|
|
72
|
+
now?: () => number
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function envFlagEnabled(): boolean {
|
|
76
|
+
const v = process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
77
|
+
return v === '1' || v === 'true'
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
function isCaptureEnabled(override: boolean | undefined): boolean {
|
|
81
|
+
if (override !== undefined) return override
|
|
82
|
+
return envFlagEnabled()
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
// One-shot "capture is live" log line, latched the first time capture
|
|
86
|
+
// actually fires (not at module import — importing this module must stay a
|
|
87
|
+
// true no-op regardless of whether the flag ends up used) — and a
|
|
88
|
+
// rate-limited write-error log so a foreign-uid EACCES (#4371 class) is
|
|
89
|
+
// observable within the window, not just discoverable from an empty corpus
|
|
90
|
+
// after the fact.
|
|
91
|
+
let loggedEnabledOnce = false
|
|
92
|
+
// `null`, not `0`: a real failure at `Date.now() === 0` (or in a test driven
|
|
93
|
+
// by `vi.setSystemTime(0)`) must still log — `0` is a valid past timestamp,
|
|
94
|
+
// not "never logged", so it cannot double as the sentinel.
|
|
95
|
+
let lastWriteErrorLoggedAt: number | null = null
|
|
96
|
+
const WRITE_ERROR_LOG_INTERVAL_MS = 5 * 60_000
|
|
97
|
+
|
|
98
|
+
function logEnabledOnce(): void {
|
|
99
|
+
if (loggedEnabledOnce) return
|
|
100
|
+
loggedEnabledOnce = true
|
|
101
|
+
try {
|
|
102
|
+
process.stderr.write(
|
|
103
|
+
`telegram gateway: speech-capture: enabled — writing $TELEGRAM_STATE_DIR/${SPEECH_CAPTURE_FILE_NAME}\n`,
|
|
104
|
+
)
|
|
105
|
+
} catch {
|
|
106
|
+
// Logging must never break the send path.
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function logWriteErrorRateLimited(err: unknown, nowMs: number): void {
|
|
111
|
+
if (lastWriteErrorLoggedAt !== null && nowMs - lastWriteErrorLoggedAt < WRITE_ERROR_LOG_INTERVAL_MS) return
|
|
112
|
+
lastWriteErrorLoggedAt = nowMs
|
|
113
|
+
try {
|
|
114
|
+
const msg = err instanceof Error ? err.message : String(err)
|
|
115
|
+
process.stderr.write(
|
|
116
|
+
`telegram gateway: speech-capture: write failed (further failures rate-limited ` +
|
|
117
|
+
`${Math.round(WRITE_ERROR_LOG_INTERVAL_MS / 60_000)}m) err=${msg}\n`,
|
|
118
|
+
)
|
|
119
|
+
} catch {
|
|
120
|
+
// Logging must never break the send path.
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* Append one `{ts, text}` JSON line to the raw-corpus capture file when
|
|
126
|
+
* capture is enabled. `text` must be the EXACT string about to be passed to
|
|
127
|
+
* `resolveVoiceOutPlan` — no trimming, no re-encoding, byte-for-byte
|
|
128
|
+
* preservation of markdown (tables, fences, pipes, emoji, backticks, ...).
|
|
129
|
+
* Off by default; a true no-op when the flag is unset. Never throws.
|
|
130
|
+
*/
|
|
131
|
+
export function captureSpeechText(text: string, opts: CaptureSpeechTextOptions = {}): void {
|
|
132
|
+
if (!isCaptureEnabled(opts.enabled)) return
|
|
133
|
+
logEnabledOnce()
|
|
134
|
+
try {
|
|
135
|
+
const stateDir = opts.stateDir ?? process.env.TELEGRAM_STATE_DIR
|
|
136
|
+
if (stateDir == null || stateDir.length === 0) return
|
|
137
|
+
const now = opts.now ?? Date.now
|
|
138
|
+
const line = `${JSON.stringify({ ts: now(), text })}\n`
|
|
139
|
+
appendFileSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME), line, {
|
|
140
|
+
encoding: 'utf8',
|
|
141
|
+
mode: SPEECH_CAPTURE_FILE_MODE,
|
|
142
|
+
})
|
|
143
|
+
} catch (err) {
|
|
144
|
+
// Capture must never break the send path — but never silently either
|
|
145
|
+
// (M2/#4371 class).
|
|
146
|
+
logWriteErrorRateLimited(err, Date.now())
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Test-only: reset the one-shot enable log and the write-error rate limiter,
|
|
152
|
+
* so a test asserting stderr output is independent of import/call order
|
|
153
|
+
* versus every other test sharing this module instance.
|
|
154
|
+
*/
|
|
155
|
+
export function __resetSpeechCaptureLogStateForTests(): void {
|
|
156
|
+
loggedEnabledOnce = false
|
|
157
|
+
lastWriteErrorLoggedAt = null
|
|
158
|
+
}
|
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
// Raw-HTML dialect handling for the Bot API 10.1 rich-markdown render path.
|
|
2
|
+
//
|
|
3
|
+
// ── Why this module exists ────────────────────────────────────────────────
|
|
4
|
+
// Agents habitually emit raw HTML tags in prose (`<b>`, `<i>`, `<a href>`).
|
|
5
|
+
// Before this module those tags reached the wire byte-verbatim, which is two
|
|
6
|
+
// distinct hazards at once:
|
|
7
|
+
//
|
|
8
|
+
// 1. Nothing establishes that Telegram's rich markdown parser accepts them.
|
|
9
|
+
// The wire-verified allowlist (raw `sendRichMessage` probes, 2026-08-13)
|
|
10
|
+
// covers exactly `<u>`, `<sub>`, `<sup>`, `<details>`/`<summary>` and
|
|
11
|
+
// `<aside>`/`<cite>`. `isParseEntitiesError` (`rich-send.ts`) matches
|
|
12
|
+
// `unsupported start tag` / `unclosed start tag`, so the wire CAN 400 on
|
|
13
|
+
// an unknown tag — and that fallback resends the body as PLAIN TEXT,
|
|
14
|
+
// where the reader then sees literal `<b>` markup.
|
|
15
|
+
// 2. mdast `html` nodes degrade to `plain` in `parse.ts`, and `plain` is
|
|
16
|
+
// `escapeMarkdown`'d on render. `escapeMarkdown` escapes `=`, so
|
|
17
|
+
// `<a href="https://example.com">` shipped as `<a href\="…">` — an
|
|
18
|
+
// attribute no parser can read.
|
|
19
|
+
//
|
|
20
|
+
// ── The policy (three buckets, degrade by TYPE, never silently) ───────────
|
|
21
|
+
// • FOLD — a tag with an exact native markdown equivalent is folded
|
|
22
|
+
// into the IR node for that construct, so it renders as the
|
|
23
|
+
// markdown the wire actually understands:
|
|
24
|
+
// <b>/<strong> -> bold **…**
|
|
25
|
+
// <i>/<em> -> italic *…*
|
|
26
|
+
// <s>/<del>/<strike> -> strike ~~…~~
|
|
27
|
+
// <code> -> code span `…`
|
|
28
|
+
// <a href="URL"> -> link [label](URL)
|
|
29
|
+
// <br> -> a line break
|
|
30
|
+
// <pre> -> fenced code block ```…```
|
|
31
|
+
// • PASSTHROUGH — the wire-verified allowlist above is emitted RAW and
|
|
32
|
+
// UNESCAPED (an IR `raw` inline), which is also what stops
|
|
33
|
+
// `escapeMarkdown` from mangling their attributes. Requires a
|
|
34
|
+
// MATCHED close: an unbalanced `<u>` emitted raw is exactly
|
|
35
|
+
// the `unclosed start tag` 400 this module exists to prevent.
|
|
36
|
+
// • DEGRADE — everything else, split by whether the token delimits
|
|
37
|
+
// content. The discriminator is MATCHING, not the tag name:
|
|
38
|
+
// – A MATCHED pair (`<marquee>…</marquee>`, `<div>…</div>`)
|
|
39
|
+
// is markup wrapped around content. The markup is dropped
|
|
40
|
+
// and the content kept; a block-level name additionally
|
|
41
|
+
// leaves a hard line break, so `<li>one</li><li>two</li>`
|
|
42
|
+
// cannot glue into `onetwo`.
|
|
43
|
+
// – A comment, or a void / self-closing tag (`<hr/>`,
|
|
44
|
+
// `<img …/>`), delimits nothing. Markup dropped; a
|
|
45
|
+
// block-level one leaves the same separator.
|
|
46
|
+
// – An UNMATCHED open/close marker is not markup at all — it
|
|
47
|
+
// is PROSE. `<service>/<key>`, `<agent>`, "the `<b>` tag",
|
|
48
|
+
// `run switchroom vault get <key>` are pervasive in this
|
|
49
|
+
// project's own agent output, and DELETING them silently
|
|
50
|
+
// mangles the agent's own reply. Such a token is kept as
|
|
51
|
+
// LITERAL TEXT with `&`/`<`/`>` HTML-entity-escaped (see
|
|
52
|
+
// `escapeHtmlLiteral`). Bot API rich markdown "can contain
|
|
53
|
+
// arbitrary HTML … parsed as described in Rich HTML
|
|
54
|
+
// style", and Rich HTML documents `<`, `>` and
|
|
55
|
+
// `&` among its supported named entities — so the
|
|
56
|
+
// entity form is the DOCUMENTED way to ship a literal
|
|
57
|
+
// angle bracket without tripping `unsupported start tag`.
|
|
58
|
+
//
|
|
59
|
+
// ── The invariant ─────────────────────────────────────────────────────────
|
|
60
|
+
// CONTENT is never lost. Only MARKUP is dropped, and only where dropping it
|
|
61
|
+
// loses no reader-visible text: a matched pair, a comment, a void tag.
|
|
62
|
+
// Anything not demonstrably markup survives to the wire as escaped literal
|
|
63
|
+
// text. (Known residual, NOT introduced by this module and not fixed here: a
|
|
64
|
+
// `<` in prose that mdast hands over as a TEXT node rather than an `html` node
|
|
65
|
+
// — the decoded `<c>`, or `response in <1s` — never reaches this module
|
|
66
|
+
// and still ships unescaped. That is a `plain`-node / `escapeMarkdown`
|
|
67
|
+
// concern.)
|
|
68
|
+
//
|
|
69
|
+
// This is a deterministic code-level guarantee, deliberately not a prompt
|
|
70
|
+
// instruction to the agents that emit the tags.
|
|
71
|
+
|
|
72
|
+
/** How a raw HTML token reads. `other` covers anything the tokenizer matched
|
|
73
|
+
* but could not classify (it degrades like an unknown tag). */
|
|
74
|
+
export type HtmlTagKind = "open" | "close" | "selfclose" | "comment" | "other";
|
|
75
|
+
|
|
76
|
+
export interface HtmlTagInfo {
|
|
77
|
+
kind: HtmlTagKind;
|
|
78
|
+
/** Lowercased tag name (`""` for a comment). */
|
|
79
|
+
name: string;
|
|
80
|
+
/** The raw source bytes of the token, verbatim. */
|
|
81
|
+
raw: string;
|
|
82
|
+
/** Raw attribute text between the tag name and the closing `>`. */
|
|
83
|
+
attrs: string;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** The IR construct a foldable tag maps onto. `pre` is BLOCK-level (a fenced
|
|
87
|
+
* code block); every other target is inline. */
|
|
88
|
+
export type HtmlFoldTarget =
|
|
89
|
+
| "bold"
|
|
90
|
+
| "italic"
|
|
91
|
+
| "strike"
|
|
92
|
+
| "code"
|
|
93
|
+
| "link"
|
|
94
|
+
| "break"
|
|
95
|
+
| "pre";
|
|
96
|
+
|
|
97
|
+
/** Tags with an exact native markdown equivalent on the Bot API 10.1 rich
|
|
98
|
+
* path. Folding them means the wire sees markdown it definitely parses
|
|
99
|
+
* instead of HTML it may reject. */
|
|
100
|
+
export const HTML_FOLD_TAGS: Readonly<Record<string, HtmlFoldTarget>> = {
|
|
101
|
+
b: "bold",
|
|
102
|
+
strong: "bold",
|
|
103
|
+
i: "italic",
|
|
104
|
+
em: "italic",
|
|
105
|
+
s: "strike",
|
|
106
|
+
del: "strike",
|
|
107
|
+
strike: "strike",
|
|
108
|
+
code: "code",
|
|
109
|
+
a: "link",
|
|
110
|
+
br: "break",
|
|
111
|
+
// `<pre>` is the one BLOCK-level fold. Telegram's own Rich HTML reference
|
|
112
|
+
// pairs `<pre><code class="language-…">` with the ```` ```lang ```` fence,
|
|
113
|
+
// so the fenced block is the exact native equivalent. Without it `<pre>`
|
|
114
|
+
// dropped its markup and the inner `<code>` folded to an INLINE span
|
|
115
|
+
// wrapping a newline — a construct Telegram will not parse, leaving the
|
|
116
|
+
// reader literal backticks around broken text.
|
|
117
|
+
pre: "pre",
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
/** HTML elements that carry no content of their own (void elements) plus the
|
|
121
|
+
* XHTML self-closing spelling. Dropping such a token's markup cannot lose
|
|
122
|
+
* text, so it degrades silently rather than surviving as literal prose.
|
|
123
|
+
* `br` is excluded — it FOLDS to a line break. */
|
|
124
|
+
export const HTML_VOID_TAGS: ReadonlySet<string> = new Set([
|
|
125
|
+
"area",
|
|
126
|
+
"base",
|
|
127
|
+
"col",
|
|
128
|
+
"embed",
|
|
129
|
+
"hr",
|
|
130
|
+
"img",
|
|
131
|
+
"input",
|
|
132
|
+
"link",
|
|
133
|
+
"meta",
|
|
134
|
+
"param",
|
|
135
|
+
"source",
|
|
136
|
+
"track",
|
|
137
|
+
"wbr",
|
|
138
|
+
]);
|
|
139
|
+
|
|
140
|
+
/** Structural / block-level element names. A degrade of one of these leaves a
|
|
141
|
+
* hard line break behind so its neighbours do not glue together
|
|
142
|
+
* (`<ul><li>one</li><li>two</li></ul>` -> `onetwo` was the defect). The
|
|
143
|
+
* passthrough allowlist (`details`, `summary`, `aside`, `cite`) is
|
|
144
|
+
* deliberately absent — those never degrade. */
|
|
145
|
+
export const HTML_BLOCK_LEVEL_TAGS: ReadonlySet<string> = new Set([
|
|
146
|
+
"address",
|
|
147
|
+
"article",
|
|
148
|
+
"blockquote",
|
|
149
|
+
"center",
|
|
150
|
+
"dd",
|
|
151
|
+
"div",
|
|
152
|
+
"dl",
|
|
153
|
+
"dt",
|
|
154
|
+
"fieldset",
|
|
155
|
+
"figcaption",
|
|
156
|
+
"figure",
|
|
157
|
+
"footer",
|
|
158
|
+
"form",
|
|
159
|
+
"h1",
|
|
160
|
+
"h2",
|
|
161
|
+
"h3",
|
|
162
|
+
"h4",
|
|
163
|
+
"h5",
|
|
164
|
+
"h6",
|
|
165
|
+
"header",
|
|
166
|
+
"hr",
|
|
167
|
+
"li",
|
|
168
|
+
"main",
|
|
169
|
+
"nav",
|
|
170
|
+
"ol",
|
|
171
|
+
"p",
|
|
172
|
+
"section",
|
|
173
|
+
"table",
|
|
174
|
+
"tbody",
|
|
175
|
+
"td",
|
|
176
|
+
"tfoot",
|
|
177
|
+
"th",
|
|
178
|
+
"thead",
|
|
179
|
+
"tr",
|
|
180
|
+
"ul",
|
|
181
|
+
]);
|
|
182
|
+
|
|
183
|
+
/** True when a degraded token should leave a line break behind. */
|
|
184
|
+
export function isBlockLevelTag(name: string): boolean {
|
|
185
|
+
return HTML_BLOCK_LEVEL_TAGS.has(name);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** True for a tag that delimits no content of its own. */
|
|
189
|
+
export function isVoidTag(name: string): boolean {
|
|
190
|
+
return HTML_VOID_TAGS.has(name);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* HTML-entity-escape a run of literal text so its angle brackets reach the
|
|
195
|
+
* reader instead of being parsed as a tag (or 400ing as an unsupported one).
|
|
196
|
+
*
|
|
197
|
+
* Bot API rich markdown "can contain arbitrary HTML … parsed as described in
|
|
198
|
+
* Rich HTML style", and Rich HTML's supported named entities are documented as
|
|
199
|
+
* exactly `< > & " ' … — –
|
|
200
|
+
* ‘ ’ “ ”` (https://core.telegram.org/bots/api). Only
|
|
201
|
+
* the first three are needed here. `&` is escaped FIRST so an already-entity
|
|
202
|
+
* -looking run cannot be double-decoded.
|
|
203
|
+
*/
|
|
204
|
+
export function escapeHtmlLiteral(text: string): string {
|
|
205
|
+
return text.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">");
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/** The `class="language-xxx"` hint on a `<pre><code …>` block, or null. */
|
|
209
|
+
const LANGUAGE_CLASS_RE = /(?:^|\s)class\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
|
|
210
|
+
|
|
211
|
+
/** Pull the fenced-block language out of a `<code class="language-python">`
|
|
212
|
+
* tag's attributes, or null. Telegram's reference spells the language hint
|
|
213
|
+
* exactly this way; anything else yields a plain fence. */
|
|
214
|
+
export function languageOf(tag: HtmlTagInfo): string | null {
|
|
215
|
+
const m = LANGUAGE_CLASS_RE.exec(tag.attrs);
|
|
216
|
+
if (m == null) return null;
|
|
217
|
+
const cls = m[1] ?? m[2] ?? m[3] ?? "";
|
|
218
|
+
const lang = cls
|
|
219
|
+
.split(/\s+/)
|
|
220
|
+
.map((c) => /^language-(.+)$/.exec(c)?.[1])
|
|
221
|
+
.find((c): c is string => c != null && c.length > 0);
|
|
222
|
+
// A fence info string cannot contain a backtick (it would close the fence).
|
|
223
|
+
return lang != null && !lang.includes("`") ? lang : null;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/** Tags PROVEN on the wire by raw `sendRichMessage` probes (2026-08-13) to
|
|
227
|
+
* render as native typed nodes. These pass through raw and unescaped.
|
|
228
|
+
* Deliberately an allowlist: an unprobed tag is not known-good syntax. */
|
|
229
|
+
export const HTML_PASSTHROUGH_TAGS: ReadonlySet<string> = new Set([
|
|
230
|
+
"u",
|
|
231
|
+
"sub",
|
|
232
|
+
"sup",
|
|
233
|
+
"details",
|
|
234
|
+
"summary",
|
|
235
|
+
"aside",
|
|
236
|
+
"cite",
|
|
237
|
+
]);
|
|
238
|
+
|
|
239
|
+
/** Matches one HTML comment or one start/end tag. Attribute values containing
|
|
240
|
+
* a raw `>` are not supported (they are invalid unquoted HTML and vanishingly
|
|
241
|
+
* rare in agent prose); such a token simply degrades. */
|
|
242
|
+
export const HTML_TOKEN_RE =
|
|
243
|
+
/<!--[\s\S]*?-->|<\/?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\/?>/g;
|
|
244
|
+
|
|
245
|
+
const TAG_RE = /^<(\/?)([A-Za-z][A-Za-z0-9-]*)((?:\s[^<>]*?)?)(\/?)>$/;
|
|
246
|
+
const HREF_RE = /(?:^|\s)href\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
|
|
247
|
+
|
|
248
|
+
/** Classify a single raw HTML token. */
|
|
249
|
+
export function classifyHtmlTag(raw: string): HtmlTagInfo {
|
|
250
|
+
if (raw.startsWith("<!--")) return { kind: "comment", name: "", raw, attrs: "" };
|
|
251
|
+
const m = TAG_RE.exec(raw);
|
|
252
|
+
if (m == null) return { kind: "other", name: "", raw, attrs: "" };
|
|
253
|
+
const [, slash, name, attrs, selfClose] = m;
|
|
254
|
+
const kind: HtmlTagKind =
|
|
255
|
+
slash === "/" ? "close" : selfClose === "/" ? "selfclose" : "open";
|
|
256
|
+
return { kind, name: name.toLowerCase(), raw, attrs: attrs ?? "" };
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/** Pull the `href` value out of a tag's attribute text, or null when absent
|
|
260
|
+
* or empty. The ORIGINAL bytes are returned — never rewritten. */
|
|
261
|
+
export function hrefOf(tag: HtmlTagInfo): string | null {
|
|
262
|
+
const m = HREF_RE.exec(tag.attrs);
|
|
263
|
+
if (m == null) return null;
|
|
264
|
+
const value = m[1] ?? m[2] ?? m[3] ?? "";
|
|
265
|
+
return value.length > 0 ? value : null;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/** True when this tag is emitted raw and unescaped (wire-verified allowlist). */
|
|
269
|
+
export function isPassthroughTag(tag: HtmlTagInfo): boolean {
|
|
270
|
+
return tag.name.length > 0 && HTML_PASSTHROUGH_TAGS.has(tag.name);
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Decide which HTML tokens in a DOCUMENT are balanced markup and which are
|
|
275
|
+
* bare prose, given every token in document order.
|
|
276
|
+
*
|
|
277
|
+
* Matching cannot be decided inside a single mdast `html` node: micromark
|
|
278
|
+
* splits `<details open>…\n\nbody\n\n</details>` into THREE nodes, so the open
|
|
279
|
+
* and close markers arrive in different streams. A document-level pass is the
|
|
280
|
+
* only place the question is answerable. Feeding it mdast `html` nodes (rather
|
|
281
|
+
* than the raw source) also means tags inside a code fence or a code span are
|
|
282
|
+
* never counted — they are not markup and must not balance anything.
|
|
283
|
+
*
|
|
284
|
+
* Standard HTML-ish matching: opens are pushed on a stack and a close pops to
|
|
285
|
+
* the nearest same-named open, leaving anything above it unmatched. Void and
|
|
286
|
+
* self-closing tags are balanced by definition and are not tracked.
|
|
287
|
+
*
|
|
288
|
+
* Returns the SOURCE OFFSETS of every token that has a partner; a token whose
|
|
289
|
+
* offset is absent is unmatched — prose, per the module policy above.
|
|
290
|
+
*/
|
|
291
|
+
export function matchedTagOffsets(
|
|
292
|
+
tokens: ReadonlyArray<{ tag: HtmlTagInfo; start: number }>,
|
|
293
|
+
): Set<number> {
|
|
294
|
+
const matched = new Set<number>();
|
|
295
|
+
const stack: { name: string; start: number }[] = [];
|
|
296
|
+
for (const { tag, start } of tokens) {
|
|
297
|
+
if (tag.kind === "open" && !isVoidTag(tag.name)) {
|
|
298
|
+
stack.push({ name: tag.name, start });
|
|
299
|
+
continue;
|
|
300
|
+
}
|
|
301
|
+
if (tag.kind !== "close") continue;
|
|
302
|
+
for (let i = stack.length - 1; i >= 0; i--) {
|
|
303
|
+
if (stack[i].name !== tag.name) continue;
|
|
304
|
+
matched.add(stack[i].start);
|
|
305
|
+
matched.add(start);
|
|
306
|
+
stack.length = i;
|
|
307
|
+
break;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
return matched;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** A piece of a raw-HTML string: either an HTML token or a run of text. */
|
|
314
|
+
export type HtmlPiece =
|
|
315
|
+
| { kind: "tag"; tag: HtmlTagInfo; start: number; end: number }
|
|
316
|
+
| { kind: "text"; text: string; start: number; end: number };
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Split a raw string into HTML tokens and the text runs between them.
|
|
320
|
+
* `base` is the absolute source offset the string starts at, so every piece
|
|
321
|
+
* carries usable UTF-16 offsets into the original markdown.
|
|
322
|
+
*/
|
|
323
|
+
export function tokenizeHtml(raw: string, base: number): HtmlPiece[] {
|
|
324
|
+
const pieces: HtmlPiece[] = [];
|
|
325
|
+
const re = new RegExp(HTML_TOKEN_RE.source, "g");
|
|
326
|
+
let last = 0;
|
|
327
|
+
let m: RegExpExecArray | null;
|
|
328
|
+
while ((m = re.exec(raw)) != null) {
|
|
329
|
+
if (m.index > last) {
|
|
330
|
+
pieces.push({
|
|
331
|
+
kind: "text",
|
|
332
|
+
text: raw.slice(last, m.index),
|
|
333
|
+
start: base + last,
|
|
334
|
+
end: base + m.index,
|
|
335
|
+
});
|
|
336
|
+
}
|
|
337
|
+
pieces.push({
|
|
338
|
+
kind: "tag",
|
|
339
|
+
tag: classifyHtmlTag(m[0]),
|
|
340
|
+
start: base + m.index,
|
|
341
|
+
end: base + m.index + m[0].length,
|
|
342
|
+
});
|
|
343
|
+
last = m.index + m[0].length;
|
|
344
|
+
}
|
|
345
|
+
if (last < raw.length) {
|
|
346
|
+
pieces.push({
|
|
347
|
+
kind: "text",
|
|
348
|
+
text: raw.slice(last),
|
|
349
|
+
start: base + last,
|
|
350
|
+
end: base + raw.length,
|
|
351
|
+
});
|
|
352
|
+
}
|
|
353
|
+
return pieces;
|
|
354
|
+
}
|