switchroom 0.21.8 → 0.21.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +100 -39
- package/dist/host-control/main.js +1 -1
- package/package.json +2 -2
- package/skills/switchroom-architecture/telegram.md +12 -10
- package/skills/switchroom-cli/SKILL.md +1 -1
- package/telegram-plugin/README.md +3 -1
- package/telegram-plugin/dist/gateway/gateway.js +919 -348
- package/telegram-plugin/format.ts +30 -5
- package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
- package/telegram-plugin/gateway/speech-capture.ts +158 -0
- package/telegram-plugin/package.json +1 -1
- package/telegram-plugin/render/code-segments.ts +38 -4
- package/telegram-plugin/render/dollar-math-guard.ts +16 -1
- package/telegram-plugin/render/html-fold.ts +354 -0
- package/telegram-plugin/render/ir.ts +53 -3
- package/telegram-plugin/render/parse.ts +642 -42
- package/telegram-plugin/render/render.ts +53 -15
- package/telegram-plugin/render/unsupported-token-guard.ts +45 -80
- package/telegram-plugin/rich-send.ts +22 -7
- package/telegram-plugin/shared/bot-runtime.ts +3 -2
- package/telegram-plugin/telegraph.ts +6 -4
- package/telegram-plugin/tests/grammy-rich-message-types.test.ts +199 -0
- package/telegram-plugin/tests/render/dollar-math-guard.test.ts +43 -0
- package/telegram-plugin/tests/render/guard-composition.test.ts +102 -0
- package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
- package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
- package/telegram-plugin/tests/render/parse.test.ts +39 -10
- package/telegram-plugin/tests/render/render.test.ts +9 -4
- package/telegram-plugin/tests/render/rich-render.test.ts +46 -5
- package/telegram-plugin/tests/render/tg-entity.test.ts +242 -0
- package/telegram-plugin/tests/render/unsupported-token-guard.test.ts +66 -66
- package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
- package/telegram-plugin/tests/sent-text-capture.test.ts +3 -3
- package/telegram-plugin/tests/speech-capture.test.ts +296 -0
- package/telegram-plugin/tests/telegraph.test.ts +1 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
- package/telegram-plugin/tts-normalize.ts +47 -9
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +17 -8
- package/telegram-plugin/voice-normalize-text.ts +48 -9
|
@@ -31,7 +31,7 @@
|
|
|
31
31
|
* - structural: gateway constructs EXACTLY ONE OutboundDedupCache and the
|
|
32
32
|
* extracted module constructs NONE (the re-`new` hard-fail rule)
|
|
33
33
|
*/
|
|
34
|
-
import { describe, it, expect } from 'vitest'
|
|
34
|
+
import { describe, it, expect, afterEach } from 'vitest'
|
|
35
35
|
import { GrammyError } from 'grammy'
|
|
36
36
|
import { readFileSync, mkdtempSync, writeFileSync } from 'node:fs'
|
|
37
37
|
import { tmpdir } from 'node:os'
|
|
@@ -58,6 +58,10 @@ import { createPendingInboundBuffer } from '../gateway/pending-inbound-buffer.js
|
|
|
58
58
|
import { SubagentReplyAuthority } from '../gateway/subagent-reply-authority.js'
|
|
59
59
|
import { redact } from '../secret-detect/redact.js'
|
|
60
60
|
import type { CurrentTurn, Access } from '../gateway/gateway.js'
|
|
61
|
+
import {
|
|
62
|
+
SPEECH_CAPTURE_FILE_NAME,
|
|
63
|
+
__resetSpeechCaptureLogStateForTests,
|
|
64
|
+
} from '../gateway/speech-capture.js'
|
|
61
65
|
|
|
62
66
|
/**
|
|
63
67
|
* Build the owner-resolution result the gateway's `resolveReplyOwnerTurn`
|
|
@@ -457,6 +461,100 @@ describe('sendReply golden harness — file sends', () => {
|
|
|
457
461
|
})
|
|
458
462
|
})
|
|
459
463
|
|
|
464
|
+
describe('speech capture (PR-0) — driven through the REAL sendReply wiring, not a synthetic call', () => {
|
|
465
|
+
const priorCapture = process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
466
|
+
const priorStateDir = process.env.TELEGRAM_STATE_DIR
|
|
467
|
+
|
|
468
|
+
afterEach(() => {
|
|
469
|
+
if (priorCapture === undefined) delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
470
|
+
else process.env.SWITCHROOM_SPEECH_CAPTURE = priorCapture
|
|
471
|
+
if (priorStateDir === undefined) delete process.env.TELEGRAM_STATE_DIR
|
|
472
|
+
else process.env.TELEGRAM_STATE_DIR = priorStateDir
|
|
473
|
+
__resetSpeechCaptureLogStateForTests()
|
|
474
|
+
})
|
|
475
|
+
|
|
476
|
+
it('M3: a capture write failure (flag ON, unwritable state dir) never breaks the reply', async () => {
|
|
477
|
+
// The destination is a FILE, not a directory, so appendFileSync fails
|
|
478
|
+
// ENOTDIR every call — the exact swallow path the flag-on window depends
|
|
479
|
+
// on for 7 days unattended. This only pins the CURRENT synchronous,
|
|
480
|
+
// in-place-caught shape: it proves a swallowed-write-failure reply still
|
|
481
|
+
// ships, but it would NOT go red if a refactor made the capture call
|
|
482
|
+
// async and fire-and-forget (unawaited) — that rejection becomes an
|
|
483
|
+
// unhandled rejection, which this codebase's policy treats as fatal
|
|
484
|
+
// (see analytics-posthog.ts, chat-lock.ts), strictly worse than the one
|
|
485
|
+
// lost capture line this test guards against, and this test alone
|
|
486
|
+
// cannot detect it. The synchronous-return contract that WOULD catch
|
|
487
|
+
// that mutation is pinned directly on `captureSpeechText` in
|
|
488
|
+
// `speech-capture.test.ts` ("G1: captureSpeechText returns undefined").
|
|
489
|
+
const dir = mkdtempSync(join(tmpdir(), 'speech-capture-golden-'))
|
|
490
|
+
const notADir = join(dir, 'state-dir-is-actually-a-file')
|
|
491
|
+
writeFileSync(notADir, 'not a directory')
|
|
492
|
+
process.env.SWITCHROOM_SPEECH_CAPTURE = '1'
|
|
493
|
+
process.env.TELEGRAM_STATE_DIR = notADir
|
|
494
|
+
|
|
495
|
+
const h = makeHarness()
|
|
496
|
+
const res = await sendReply(h.deps, req('The reply must still ship. ✅ done.'))
|
|
497
|
+
const sends = h.calls.filter((c) => c.method === 'sendRichMessage')
|
|
498
|
+
expect(sends).toHaveLength(1)
|
|
499
|
+
expect(sends[0]!.text).toContain('The reply must still ship.')
|
|
500
|
+
expect(res.content[0]!.text).toMatch(/^sent \(id: \d+\)$/)
|
|
501
|
+
})
|
|
502
|
+
|
|
503
|
+
it('M3: the captured string IS the exact argument passed to resolveVoiceOutPlan — not merely textually adjacent', async () => {
|
|
504
|
+
const stateDir = mkdtempSync(join(tmpdir(), 'speech-capture-golden-'))
|
|
505
|
+
process.env.SWITCHROOM_SPEECH_CAPTURE = '1'
|
|
506
|
+
process.env.TELEGRAM_STATE_DIR = stateDir
|
|
507
|
+
|
|
508
|
+
const h = makeHarness()
|
|
509
|
+
let observedByVoicePlan: string | null = null
|
|
510
|
+
h.deps.resolveVoiceOutPlan = (_voiceOut, replyText) => {
|
|
511
|
+
observedByVoicePlan = replyText
|
|
512
|
+
return null
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
// Deliberately includes a spaced em-dash: `normalizeOutboundBody`'s
|
|
516
|
+
// punctuation/bold pass (`normalizePunctuation`, format.ts:695, step 4
|
|
517
|
+
// of the pipeline — outbound-send-path.ts's own docstring: "normalize →
|
|
518
|
+
// redact → punctuation/bold → voice-scrub") rewrites a bare/mid-prose
|
|
519
|
+
// em-dash into `, `. A golden fixture with nothing for the pipeline to
|
|
520
|
+
// transform can't tell "capture reads the normalised `text`" apart from
|
|
521
|
+
// "capture reads the raw `rawText`" — both would produce byte-identical
|
|
522
|
+
// output. This fixture makes that distinction observable (G2).
|
|
523
|
+
const raw = '| A | B |\n| --- | --- |\n| 1 | 2 |\n\n```ts\nconst x = 1 | 2\n```\n\nStatus update — done. ✅ `code` 😀.'
|
|
524
|
+
await sendReply(h.deps, req(raw))
|
|
525
|
+
|
|
526
|
+
expect(observedByVoicePlan).not.toBeNull()
|
|
527
|
+
const captured = readFileSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME), 'utf8')
|
|
528
|
+
.split('\n')
|
|
529
|
+
.filter((l) => l.length > 0)
|
|
530
|
+
.map((l) => JSON.parse(l) as { ts: number; text: string })
|
|
531
|
+
expect(captured).toHaveLength(1)
|
|
532
|
+
// The load-bearing assertion: the capture is not "close to" or "derived
|
|
533
|
+
// from" resolveVoiceOutPlan's input — it is the SAME string, by identity
|
|
534
|
+
// of value, not by coincidence of two independent pipelines agreeing.
|
|
535
|
+
expect(captured[0]!.text).toBe(observedByVoicePlan)
|
|
536
|
+
// G2: the captured text is the NORMALISED form, downstream of
|
|
537
|
+
// `normalizePunctuation`'s em-dash rewrite — not the raw fixture. A
|
|
538
|
+
// regression that swapped in the pre-pipeline `rawText` would still
|
|
539
|
+
// satisfy the identity assertion above (both call sites would agree on
|
|
540
|
+
// SOME string), so this must check the captured content against the
|
|
541
|
+
// known transform independently, not just against `observedByVoicePlan`.
|
|
542
|
+
expect(captured[0]!.text).not.toBe(raw)
|
|
543
|
+
expect(captured[0]!.text).not.toContain('Status update — done')
|
|
544
|
+
expect(captured[0]!.text).toContain('Status update, done')
|
|
545
|
+
})
|
|
546
|
+
|
|
547
|
+
it('flag OFF: sendReply never creates the capture file, even with a writable state dir', async () => {
|
|
548
|
+
const stateDir = mkdtempSync(join(tmpdir(), 'speech-capture-golden-'))
|
|
549
|
+
delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
550
|
+
process.env.TELEGRAM_STATE_DIR = stateDir
|
|
551
|
+
|
|
552
|
+
const h = makeHarness()
|
|
553
|
+
await sendReply(h.deps, req('Never captured.'))
|
|
554
|
+
expect(() => readFileSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME), 'utf8')).toThrow()
|
|
555
|
+
})
|
|
556
|
+
})
|
|
557
|
+
|
|
460
558
|
describe('cross-surface dedup — the ONE OutboundDedupCache instance (Amendment 1/9)', () => {
|
|
461
559
|
it('a stream-surface record suppresses the same-content reply (zero wire calls)', async () => {
|
|
462
560
|
const dedup = new OutboundDedupCache()
|
|
@@ -43,9 +43,9 @@ const CHAT = 5550001
|
|
|
43
43
|
/**
|
|
44
44
|
* The response Telegram actually returns for `sendRichMessage` — the body lives
|
|
45
45
|
* in `rich_message.blocks`; `text` and `caption` are ABSENT. Verified against
|
|
46
|
-
* `@grammyjs/types`
|
|
47
|
-
* `message.d.ts:
|
|
48
|
-
* MsgWith<"rich_message">`) and `:
|
|
46
|
+
* `@grammyjs/types` 4.0.0 (the version `grammy@^1.45` resolves)
|
|
47
|
+
* `message.d.ts:98` (`RichMessageMessage = CommonMessage &
|
|
48
|
+
* MsgWith<"rich_message">`) and `:184` (`rich_message?: RichMessage`), and
|
|
49
49
|
* against the Bot API reference: `sendRichMessage` "On success, the sent
|
|
50
50
|
* Message is returned", `Message.rich_message: RichMessage` "Optional. Message
|
|
51
51
|
* is a rich formatted message", `RichMessage.blocks` "Content of the message".
|
|
@@ -0,0 +1,296 @@
|
|
|
1
|
+
import { describe, it, expect, afterEach, vi } from 'vitest'
|
|
2
|
+
import { chmodSync, existsSync, mkdtempSync, mkdirSync, readFileSync, rmSync, statSync, writeFileSync } from 'node:fs'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import {
|
|
6
|
+
captureSpeechText,
|
|
7
|
+
SPEECH_CAPTURE_FILE_NAME,
|
|
8
|
+
__resetSpeechCaptureLogStateForTests,
|
|
9
|
+
} from '../gateway/speech-capture.js'
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* PR-0 (`/tmp/claude-0/tts/out/fable-plan-v2.md` §9): flag-gated raw-markdown
|
|
13
|
+
* capture hook. Every test here points at an explicit `stateDir` override —
|
|
14
|
+
* never at `process.env.TELEGRAM_STATE_DIR` — so nothing in this suite can
|
|
15
|
+
* touch a live agent's state tree (`TELEGRAM_STATE_DIR` is not covered by the
|
|
16
|
+
* repo's agent-state-dir hermeticity guard, see
|
|
17
|
+
* `tests/vitest-setup/agent-state-dir-guard-core.mjs`).
|
|
18
|
+
*
|
|
19
|
+
* The integration-level guarantees (a write failure never breaks the REAL
|
|
20
|
+
* `sendReply`; the captured string IS the exact `resolveVoiceOutPlan`
|
|
21
|
+
* argument, not merely textually adjacent) are pinned separately in
|
|
22
|
+
* `send-reply-golden.test.ts`, driven through the real wiring — this file
|
|
23
|
+
* covers the module's own contract in isolation.
|
|
24
|
+
*/
|
|
25
|
+
describe('captureSpeechText', () => {
|
|
26
|
+
const dirs: string[] = []
|
|
27
|
+
function freshDir(): string {
|
|
28
|
+
const d = mkdtempSync(join(tmpdir(), 'speech-capture-test-'))
|
|
29
|
+
dirs.push(d)
|
|
30
|
+
return d
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
afterEach(() => {
|
|
34
|
+
for (const d of dirs.splice(0)) {
|
|
35
|
+
rmSync(d, { recursive: true, force: true })
|
|
36
|
+
}
|
|
37
|
+
__resetSpeechCaptureLogStateForTests()
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
it('flag off (explicit enabled:false) writes nothing — true no-op', () => {
|
|
41
|
+
const stateDir = freshDir()
|
|
42
|
+
captureSpeechText('hello world', { enabled: false, stateDir })
|
|
43
|
+
expect(existsSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME))).toBe(false)
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
it('G1: captureSpeechText returns undefined (sync contract), flag on and flag off alike', () => {
|
|
47
|
+
// Pins the SYNCHRONOUS return contract the swallow-on-write-failure
|
|
48
|
+
// guarantee depends on. `send-reply-golden.test.ts`'s M3 "write failure
|
|
49
|
+
// never breaks the reply" test only observes that the reply still ships
|
|
50
|
+
// — it would NOT go red if a refactor made this function `async` and
|
|
51
|
+
// fire-and-forget (an unawaited rejection there becomes an unhandled
|
|
52
|
+
// rejection, which this codebase's policy treats as fatal — see
|
|
53
|
+
// analytics-posthog.ts and chat-lock.ts). An `async function` always
|
|
54
|
+
// returns a Promise, never `undefined`, so THIS assertion is the one
|
|
55
|
+
// that goes red on that mutation, awaited or not.
|
|
56
|
+
const stateDir = freshDir()
|
|
57
|
+
expect(captureSpeechText('x', { enabled: false, stateDir })).toBeUndefined()
|
|
58
|
+
expect(captureSpeechText('x', { enabled: true, stateDir })).toBeUndefined()
|
|
59
|
+
// Also true on the swallowed-write-failure path (destination collides
|
|
60
|
+
// with a file, not a dir) — the exact shape M3's golden test exercises.
|
|
61
|
+
const notADir = join(freshDir(), 'not-a-directory-marker')
|
|
62
|
+
writeFileSync(notADir, 'not a directory')
|
|
63
|
+
expect(captureSpeechText('x', { enabled: true, stateDir: notADir })).toBeUndefined()
|
|
64
|
+
})
|
|
65
|
+
|
|
66
|
+
it('flag off via env (SWITCHROOM_SPEECH_CAPTURE unset) writes nothing', () => {
|
|
67
|
+
const stateDir = freshDir()
|
|
68
|
+
const prior = process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
69
|
+
delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
70
|
+
try {
|
|
71
|
+
captureSpeechText('hello world', { stateDir })
|
|
72
|
+
expect(existsSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME))).toBe(false)
|
|
73
|
+
} finally {
|
|
74
|
+
if (prior === undefined) delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
75
|
+
else process.env.SWITCHROOM_SPEECH_CAPTURE = prior
|
|
76
|
+
}
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
it('flag off: never touches the filesystem, even with a garbage stateDir', () => {
|
|
80
|
+
// Negligible-cost proof: an unwritable/garbage destination must not
|
|
81
|
+
// surface any error or side effect when the flag is off — the function
|
|
82
|
+
// should return before touching fs at all.
|
|
83
|
+
expect(() => captureSpeechText('x', {
|
|
84
|
+
enabled: false,
|
|
85
|
+
stateDir: '/definitely/does/not/exist/\0bad',
|
|
86
|
+
})).not.toThrow()
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
it('M1: flag accepts "1"', () => {
|
|
90
|
+
const stateDir = freshDir()
|
|
91
|
+
const prior = process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
92
|
+
process.env.SWITCHROOM_SPEECH_CAPTURE = '1'
|
|
93
|
+
try {
|
|
94
|
+
captureSpeechText('x', { stateDir })
|
|
95
|
+
expect(existsSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME))).toBe(true)
|
|
96
|
+
} finally {
|
|
97
|
+
if (prior === undefined) delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
98
|
+
else process.env.SWITCHROOM_SPEECH_CAPTURE = prior
|
|
99
|
+
}
|
|
100
|
+
})
|
|
101
|
+
|
|
102
|
+
it('M1: flag accepts "true" (text-voice-scrub.ts:102 convention) — not silently a no-op', () => {
|
|
103
|
+
const stateDir = freshDir()
|
|
104
|
+
const prior = process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
105
|
+
process.env.SWITCHROOM_SPEECH_CAPTURE = 'true'
|
|
106
|
+
try {
|
|
107
|
+
captureSpeechText('x', { stateDir })
|
|
108
|
+
expect(existsSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME))).toBe(true)
|
|
109
|
+
} finally {
|
|
110
|
+
if (prior === undefined) delete process.env.SWITCHROOM_SPEECH_CAPTURE
|
|
111
|
+
else process.env.SWITCHROOM_SPEECH_CAPTURE = prior
|
|
112
|
+
}
|
|
113
|
+
})
|
|
114
|
+
|
|
115
|
+
it('flag on appends the exact string, byte-for-byte, including a table, a fenced code block, an emoji, and inline backticks', () => {
|
|
116
|
+
const stateDir = freshDir()
|
|
117
|
+
const raw = [
|
|
118
|
+
'# Heading',
|
|
119
|
+
'',
|
|
120
|
+
'| A | B |',
|
|
121
|
+
'| --- | --- |',
|
|
122
|
+
'| 1 | 2 |',
|
|
123
|
+
'',
|
|
124
|
+
'```ts',
|
|
125
|
+
'const x = 1 | 2',
|
|
126
|
+
'```',
|
|
127
|
+
'',
|
|
128
|
+
'Status: ✅ done, use `inline code` with a pipe | here, and an emoji 😀.',
|
|
129
|
+
].join('\n')
|
|
130
|
+
captureSpeechText(raw, { enabled: true, stateDir, now: () => 1234567890 })
|
|
131
|
+
const filePath = join(stateDir, SPEECH_CAPTURE_FILE_NAME)
|
|
132
|
+
expect(existsSync(filePath)).toBe(true)
|
|
133
|
+
const contents = readFileSync(filePath, 'utf8')
|
|
134
|
+
const lines = contents.split('\n').filter((l) => l.length > 0)
|
|
135
|
+
expect(lines).toHaveLength(1)
|
|
136
|
+
const parsed = JSON.parse(lines[0]) as { ts: number; text: string }
|
|
137
|
+
expect(parsed.ts).toBe(1234567890)
|
|
138
|
+
// Byte-for-byte: exact equality, not a fuzzy/normalized comparison.
|
|
139
|
+
expect(parsed.text).toBe(raw)
|
|
140
|
+
// Explicitly pin the bytes the redesign depends on surviving capture.
|
|
141
|
+
expect(parsed.text).toContain('| A | B |')
|
|
142
|
+
expect(parsed.text).toContain('```ts')
|
|
143
|
+
expect(parsed.text).toContain('`inline code`')
|
|
144
|
+
expect(parsed.text).toContain('✅')
|
|
145
|
+
expect(parsed.text).toContain('😀')
|
|
146
|
+
})
|
|
147
|
+
|
|
148
|
+
it('flag on preserves a lone (unpaired) surrogate byte-for-byte through the JSON round-trip', () => {
|
|
149
|
+
const stateDir = freshDir()
|
|
150
|
+
// A lone high surrogate with no low-surrogate partner — the kind of
|
|
151
|
+
// malformed-but-real string content a JS string can hold (e.g. a
|
|
152
|
+
// truncated emoji from an upstream mangling) that a naive re-encoding
|
|
153
|
+
// pass can silently mutate or drop.
|
|
154
|
+
const raw = 'before \uD800 after'
|
|
155
|
+
captureSpeechText(raw, { enabled: true, stateDir, now: () => 1 })
|
|
156
|
+
const filePath = join(stateDir, SPEECH_CAPTURE_FILE_NAME)
|
|
157
|
+
const line = readFileSync(filePath, 'utf8').split('\n').filter((l) => l.length > 0)[0]!
|
|
158
|
+
const parsed = JSON.parse(line) as { ts: number; text: string }
|
|
159
|
+
expect(parsed.text).toBe(raw)
|
|
160
|
+
expect(parsed.text.charCodeAt(parsed.text.indexOf('\uD800'))).toBe(0xd800)
|
|
161
|
+
})
|
|
162
|
+
|
|
163
|
+
it('flag on appends one JSON line per call, in order', () => {
|
|
164
|
+
const stateDir = freshDir()
|
|
165
|
+
captureSpeechText('first', { enabled: true, stateDir, now: () => 1 })
|
|
166
|
+
captureSpeechText('second', { enabled: true, stateDir, now: () => 2 })
|
|
167
|
+
const filePath = join(stateDir, SPEECH_CAPTURE_FILE_NAME)
|
|
168
|
+
const lines = readFileSync(filePath, 'utf8').split('\n').filter((l) => l.length > 0)
|
|
169
|
+
expect(lines).toHaveLength(2)
|
|
170
|
+
expect(JSON.parse(lines[0])).toEqual({ ts: 1, text: 'first' })
|
|
171
|
+
expect(JSON.parse(lines[1])).toEqual({ ts: 2, text: 'second' })
|
|
172
|
+
})
|
|
173
|
+
|
|
174
|
+
it('flag on but stateDir unresolved (no TELEGRAM_STATE_DIR, no override) writes nothing and does not throw', () => {
|
|
175
|
+
const prior = process.env.TELEGRAM_STATE_DIR
|
|
176
|
+
delete process.env.TELEGRAM_STATE_DIR
|
|
177
|
+
try {
|
|
178
|
+
expect(() => captureSpeechText('x', { enabled: true })).not.toThrow()
|
|
179
|
+
} finally {
|
|
180
|
+
if (prior === undefined) delete process.env.TELEGRAM_STATE_DIR
|
|
181
|
+
else process.env.TELEGRAM_STATE_DIR = prior
|
|
182
|
+
}
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
it('B1: the capture file is created 0o600 (owner-only), not the umask default', () => {
|
|
186
|
+
const stateDir = freshDir()
|
|
187
|
+
captureSpeechText('x', { enabled: true, stateDir })
|
|
188
|
+
const filePath = join(stateDir, SPEECH_CAPTURE_FILE_NAME)
|
|
189
|
+
const mode = statSync(filePath).mode & 0o777
|
|
190
|
+
expect(mode).toBe(0o600)
|
|
191
|
+
})
|
|
192
|
+
|
|
193
|
+
it('B1: mode stays 0o600 across repeated appends to an already-created file', () => {
|
|
194
|
+
const stateDir = freshDir()
|
|
195
|
+
captureSpeechText('first', { enabled: true, stateDir })
|
|
196
|
+
// Second call appends to the now-existing file; mode must not have been
|
|
197
|
+
// relaxed by any subsequent create-mode default.
|
|
198
|
+
captureSpeechText('second', { enabled: true, stateDir })
|
|
199
|
+
const filePath = join(stateDir, SPEECH_CAPTURE_FILE_NAME)
|
|
200
|
+
const mode = statSync(filePath).mode & 0o777
|
|
201
|
+
expect(mode).toBe(0o600)
|
|
202
|
+
})
|
|
203
|
+
|
|
204
|
+
it('B1/no dead mkdir: a stateDir that does not exist is a swallowed write failure, not silently created', () => {
|
|
205
|
+
// The prior implementation's `mkdirSync(stateDir, {recursive:true, mode:
|
|
206
|
+
// 0o700})` protected nothing in production (TELEGRAM_STATE_DIR always
|
|
207
|
+
// pre-exists) and is gone; a genuinely-missing dir now fails the write
|
|
208
|
+
// (ENOENT) exactly like any other write-failure shape, swallowed.
|
|
209
|
+
const parent = freshDir()
|
|
210
|
+
const missing = join(parent, 'does', 'not', 'exist')
|
|
211
|
+
expect(() => captureSpeechText('x', { enabled: true, stateDir: missing })).not.toThrow()
|
|
212
|
+
expect(existsSync(missing)).toBe(false)
|
|
213
|
+
})
|
|
214
|
+
|
|
215
|
+
it('a write error (destination path collides with a file, not a dir) does not throw', () => {
|
|
216
|
+
const parent = freshDir()
|
|
217
|
+
const stateDir = join(parent, 'not-a-directory')
|
|
218
|
+
writeFileSync(stateDir, 'not a directory')
|
|
219
|
+
expect(() => captureSpeechText('should not throw', { enabled: true, stateDir })).not.toThrow()
|
|
220
|
+
expect(existsSync(join(stateDir, SPEECH_CAPTURE_FILE_NAME))).toBe(false)
|
|
221
|
+
})
|
|
222
|
+
|
|
223
|
+
it('a write error into a read-only directory does not throw', () => {
|
|
224
|
+
const stateDir = freshDir()
|
|
225
|
+
mkdirSync(stateDir, { recursive: true })
|
|
226
|
+
const originalMode = 0o755
|
|
227
|
+
try {
|
|
228
|
+
// Some sandboxes ignore chmod as non-root; this test still asserts the
|
|
229
|
+
// no-throw contract either way, it just may not exercise the EACCES
|
|
230
|
+
// path when running as root.
|
|
231
|
+
chmodSync(stateDir, 0o500)
|
|
232
|
+
expect(() => captureSpeechText('x', { enabled: true, stateDir })).not.toThrow()
|
|
233
|
+
} finally {
|
|
234
|
+
chmodSync(stateDir, originalMode)
|
|
235
|
+
}
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
describe('M2: observability — never silent for the full 7-day window', () => {
|
|
239
|
+
it('logs one stderr line the first time capture actually fires, not on every call', () => {
|
|
240
|
+
const stateDir = freshDir()
|
|
241
|
+
const spy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true)
|
|
242
|
+
try {
|
|
243
|
+
captureSpeechText('first', { enabled: true, stateDir })
|
|
244
|
+
captureSpeechText('second', { enabled: true, stateDir })
|
|
245
|
+
captureSpeechText('third', { enabled: true, stateDir })
|
|
246
|
+
const enableLines = spy.mock.calls.filter((c) => String(c[0]).includes('speech-capture: enabled'))
|
|
247
|
+
expect(enableLines).toHaveLength(1)
|
|
248
|
+
} finally {
|
|
249
|
+
spy.mockRestore()
|
|
250
|
+
}
|
|
251
|
+
})
|
|
252
|
+
|
|
253
|
+
it('does not log the "enabled" line when the flag is off', () => {
|
|
254
|
+
const stateDir = freshDir()
|
|
255
|
+
const spy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true)
|
|
256
|
+
try {
|
|
257
|
+
captureSpeechText('x', { enabled: false, stateDir })
|
|
258
|
+
const enableLines = spy.mock.calls.filter((c) => String(c[0]).includes('speech-capture: enabled'))
|
|
259
|
+
expect(enableLines).toHaveLength(0)
|
|
260
|
+
} finally {
|
|
261
|
+
spy.mockRestore()
|
|
262
|
+
}
|
|
263
|
+
})
|
|
264
|
+
|
|
265
|
+
it('logs a write-failure line on the first failure, then rate-limits further failures', () => {
|
|
266
|
+
// Manual Date.now mocking, NOT vi.useFakeTimers()/vi.setSystemTime():
|
|
267
|
+
// this file runs under both vitest AND `bun test` (telegram-plugin's
|
|
268
|
+
// bun-test-ci.sh substring-matches the whole tests/ dir), and bun's
|
|
269
|
+
// vitest shim does not implement `vi.setSystemTime` — see
|
|
270
|
+
// progress-update.test.ts's note on the same gotcha (CI build #48).
|
|
271
|
+
// `vi.spyOn` + `mockImplementation` works under both runners.
|
|
272
|
+
let mockNow = 0
|
|
273
|
+
const dateSpy = vi.spyOn(Date, 'now').mockImplementation(() => mockNow)
|
|
274
|
+
const parent = freshDir()
|
|
275
|
+
const stateDir = join(parent, 'not-a-directory')
|
|
276
|
+
writeFileSync(stateDir, 'not a directory')
|
|
277
|
+
const spy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true)
|
|
278
|
+
try {
|
|
279
|
+
captureSpeechText('one', { enabled: true, stateDir })
|
|
280
|
+
captureSpeechText('two', { enabled: true, stateDir })
|
|
281
|
+
mockNow = 1000 // 1s later — still inside the rate-limit window
|
|
282
|
+
captureSpeechText('three', { enabled: true, stateDir })
|
|
283
|
+
const failLines = spy.mock.calls.filter((c) => String(c[0]).includes('speech-capture: write failed'))
|
|
284
|
+
expect(failLines).toHaveLength(1)
|
|
285
|
+
|
|
286
|
+
mockNow = 6 * 60_000 // past the 5-minute window
|
|
287
|
+
captureSpeechText('four', { enabled: true, stateDir })
|
|
288
|
+
const failLinesAfter = spy.mock.calls.filter((c) => String(c[0]).includes('speech-capture: write failed'))
|
|
289
|
+
expect(failLinesAfter).toHaveLength(2)
|
|
290
|
+
} finally {
|
|
291
|
+
spy.mockRestore()
|
|
292
|
+
dateSpy.mockRestore()
|
|
293
|
+
}
|
|
294
|
+
})
|
|
295
|
+
})
|
|
296
|
+
})
|
|
@@ -165,7 +165,7 @@ describe('markdownToTelegraphNodes — block elements', () => {
|
|
|
165
165
|
})
|
|
166
166
|
|
|
167
167
|
it('folds an expandable `**>` quote into one blockquote, marker stripped', () => {
|
|
168
|
-
// The
|
|
168
|
+
// The LEGACY switchroom expandable-quote encoding: `**>` opener on the first
|
|
169
169
|
// line, `>` continuation lines. Telegra.ph has no collapsible-blockquote
|
|
170
170
|
// tag, so it degrades to a normal blockquote — but the `**>` marker MUST be
|
|
171
171
|
// recognised. Before the fix the `**>` line became a paragraph with literal
|
|
@@ -184,6 +184,120 @@ describe('numbers', () => {
|
|
|
184
184
|
'rename class5 and my5thing',
|
|
185
185
|
)
|
|
186
186
|
})
|
|
187
|
+
|
|
188
|
+
// Regression: unit-suffix lookbehind + case sensitivity. Corpus in
|
|
189
|
+
// /tmp/claude-0/tts/corpus.json (1,679 samples of ALREADY-NORMALIZED
|
|
190
|
+
// production output, not raw input): 146 samples already carry the
|
|
191
|
+
// shipped mangle artifact this fix removes, 53 carry an unfixed decimal/
|
|
192
|
+
// comma+unit token (see the "known gap" test below), and 61 samples
|
|
193
|
+
// change under this fix. A decimal seconds value like "0.17s" was matched
|
|
194
|
+
// by \b re-anchoring between the "." and the following digits, so "17s"
|
|
195
|
+
// alone got read as "seventeen seconds" and split the decimal apart; and
|
|
196
|
+
// the /i flag let a bare capital letter (money shorthand "5M", a memory
|
|
197
|
+
// size "4G", a product name "4080S") match a single-letter unit
|
|
198
|
+
// alternative it was never meant to.
|
|
199
|
+
test('decimal seconds survive intact — no mid-decimal unit match', () => {
|
|
200
|
+
expect(normalizeForTts('latency was 0.17s')).toBe('latency was 0.17s')
|
|
201
|
+
expect(normalizeForTts('took 1.5s to load')).toBe('took 1.5s to load')
|
|
202
|
+
expect(normalizeForTts('waited 2.25s')).toBe('waited 2.25s')
|
|
203
|
+
expect(normalizeForTts('under 0.9s')).toBe('under 0.9s')
|
|
204
|
+
})
|
|
205
|
+
|
|
206
|
+
// KNOWN GAP (disclosed, not fixed in this hotfix — see the comment above
|
|
207
|
+
// the unit-suffix regex in tts-normalize.ts): a decimal- or comma-glued
|
|
208
|
+
// number+unit token is now left completely unexpanded (unit unspoken)
|
|
209
|
+
// rather than mangled. This is a real, measured trade-off — corpus
|
|
210
|
+
// samples carrying an unspoken digit-adjacent unit go 65 → 122 (61
|
|
211
|
+
// samples change, every one gaining an unspoken unit) — pinned here as
|
|
212
|
+
// the baseline a future decimal-expansion pass should move.
|
|
213
|
+
test('KNOWN GAP: decimal/comma-glued number+unit is left unexpanded, not mangled', () => {
|
|
214
|
+
expect(normalizeForTts('free: 13.0 GB')).toBe('free: 13.0 GB')
|
|
215
|
+
expect(normalizeForTts('weighs 89.5g')).toBe('weighs 89.5g')
|
|
216
|
+
expect(normalizeForTts('ran for 27.5 s')).toBe('ran for 27.5 s')
|
|
217
|
+
expect(normalizeForTts('paused 9.5 min')).toBe('paused 9.5 min')
|
|
218
|
+
expect(normalizeForTts('spread over 0.7-0.8 m')).toBe('spread over 0.7-0.8 m')
|
|
219
|
+
})
|
|
220
|
+
|
|
221
|
+
// MINOR: comma leak — same defect class as the decimal lookbehind, closed
|
|
222
|
+
// in the same lookbehind by also excluding `,`. Before this: "1,500s" →
|
|
223
|
+
// "1,five hundred seconds", "12,000s" → "12,zero seconds", "1,024MB" →
|
|
224
|
+
// "1,twenty-four megabytes". The lookbehind is adjacency-only so an
|
|
225
|
+
// ordinary sentence comma ("wait, 5s") still expands correctly — only a
|
|
226
|
+
// digit glued directly to the comma is blocked.
|
|
227
|
+
test('comma-glued numbers are left unmangled; a sentence comma still expands', () => {
|
|
228
|
+
expect(normalizeForTts('rate limited: 1,500s')).toBe('rate limited: 1,500s')
|
|
229
|
+
expect(normalizeForTts('timeout after 12,000s')).toBe('timeout after 12,000s')
|
|
230
|
+
expect(normalizeForTts('buffer is 1,024MB')).toBe('buffer is 1,024MB')
|
|
231
|
+
expect(normalizeForTts('wait, 5s')).toBe('wait, five seconds')
|
|
232
|
+
})
|
|
233
|
+
|
|
234
|
+
test('whole-number seconds still expand: 90s → ninety seconds', () => {
|
|
235
|
+
expect(normalizeForTts('done in 90s')).toBe('done in ninety seconds')
|
|
236
|
+
})
|
|
237
|
+
|
|
238
|
+
test('units unaffected by the case-sensitivity fix: 500ms, 5MB, 2GB, 16 m', () => {
|
|
239
|
+
expect(normalizeForTts('took 500ms')).toBe('took five hundred milliseconds')
|
|
240
|
+
expect(normalizeForTts('file is 5MB')).toBe('file is five megabytes')
|
|
241
|
+
expect(normalizeForTts('disk has 2GB')).toBe('disk has two gigabytes')
|
|
242
|
+
expect(normalizeForTts('ran 16 m')).toBe('ran sixteen metres')
|
|
243
|
+
})
|
|
244
|
+
|
|
245
|
+
test('bare capital M never reads as minutes/metres (money shorthand, corpus)', () => {
|
|
246
|
+
expect(normalizeForTts('Revenue was 5M this year')).toBe(
|
|
247
|
+
'Revenue was 5M this year',
|
|
248
|
+
)
|
|
249
|
+
// Pre-existing quirk, not introduced or fixed here: the currency regex
|
|
250
|
+
// consumes "$5" before the unit pass ever sees the dangling "M", so this
|
|
251
|
+
// never read as "minutes" even on unpatched main — pin the ACTUAL output
|
|
252
|
+
// instead of an assertion that can't fail in either direction.
|
|
253
|
+
expect(normalizeForTts('raised $5M in the round')).toBe(
|
|
254
|
+
'raised five dollarsM in the round',
|
|
255
|
+
)
|
|
256
|
+
})
|
|
257
|
+
|
|
258
|
+
test('HTTP-status-shaped input adjacent to a capital letter is not read as a duration', () => {
|
|
259
|
+
expect(normalizeForTts('a 503M error occurred')).toBe(
|
|
260
|
+
'a 503M error occurred',
|
|
261
|
+
)
|
|
262
|
+
expect(normalizeForTts('status 503M')).toBe('status 503M')
|
|
263
|
+
})
|
|
264
|
+
|
|
265
|
+
// Corpus wins from the case-sensitivity split (22 samples improved, 14 of
|
|
266
|
+
// them "G" memory sizes wrongly read as GRAMS, plus an RTX "4080S" wrongly
|
|
267
|
+
// read as a duration). These already passed after the original fix but
|
|
268
|
+
// were previously unpinned.
|
|
269
|
+
test('single-letter units case-split: bare capitals stay untouched (was misread pre-fix)', () => {
|
|
270
|
+
expect(normalizeForTts('done in 5S')).toBe('done in 5S')
|
|
271
|
+
expect(normalizeForTts('wait 3H')).toBe('wait 3H')
|
|
272
|
+
expect(normalizeForTts('ships in 10D')).toBe('ships in 10D')
|
|
273
|
+
expect(normalizeForTts('needs 2G')).toBe('needs 2G')
|
|
274
|
+
expect(normalizeForTts('ran 16 M')).toBe('ran 16 M')
|
|
275
|
+
})
|
|
276
|
+
|
|
277
|
+
test('memory-size "G" no longer read as grams (corpus: 4G/12G/97G)', () => {
|
|
278
|
+
expect(normalizeForTts('upgrade to 4G')).toBe('upgrade to 4G')
|
|
279
|
+
expect(normalizeForTts('needs 12G')).toBe('needs 12G')
|
|
280
|
+
expect(normalizeForTts('disk has 97G')).toBe('disk has 97G')
|
|
281
|
+
})
|
|
282
|
+
|
|
283
|
+
test('product name "4080S" (RTX 4080 Super) not read as a duration', () => {
|
|
284
|
+
expect(normalizeForTts('bought a 4080S')).toBe('bought a 4080S')
|
|
285
|
+
})
|
|
286
|
+
})
|
|
287
|
+
|
|
288
|
+
describe('composed pipeline: normalizeForTts(normalizeForSpeech(x))', () => {
|
|
289
|
+
// Fleet-critical, previously unguarded: production always composes pass 1
|
|
290
|
+
// (voice-normalize-text.ts, NO \s? between number and unit, "m" → minute)
|
|
291
|
+
// with pass 2 (tts-normalize.ts, optional \s?, "m" → metre). The
|
|
292
|
+
// minute-vs-metre reading is an EMERGENT property of that composition, not
|
|
293
|
+
// a property either pass exhibits alone.
|
|
294
|
+
test('glued "5m" reads as minutes (pass 1 claims it first, no space)', () => {
|
|
295
|
+
expect(normalizeForTts(normalizeForSpeech('run 5m'))).toBe('run five minutes')
|
|
296
|
+
})
|
|
297
|
+
|
|
298
|
+
test('spaced "16 m" reads as metres (pass 1 has no \\s?, so pass 2 claims it)', () => {
|
|
299
|
+
expect(normalizeForTts(normalizeForSpeech('ran 16 m'))).toBe('ran sixteen metres')
|
|
300
|
+
})
|
|
187
301
|
})
|
|
188
302
|
|
|
189
303
|
describe('URLs', () => {
|
|
@@ -193,6 +193,95 @@ describe('normalizeForSpeech — numbers, units & symbols', () => {
|
|
|
193
193
|
expect(normalizeForSpeech('my5thing works')).toBe('my5thing works')
|
|
194
194
|
expect(normalizeForSpeech('class5 room')).toBe('class5 room')
|
|
195
195
|
})
|
|
196
|
+
|
|
197
|
+
// Regression: unit-suffix lookbehind + case sensitivity. Corpus in
|
|
198
|
+
// /tmp/claude-0/tts/corpus.json (1,679 samples of ALREADY-NORMALIZED
|
|
199
|
+
// production output, not raw input): 146 samples already carry the
|
|
200
|
+
// shipped mangle artifact this fix removes, 53 carry an unfixed decimal/
|
|
201
|
+
// comma+unit token (see the "known gap" test below), and 61 samples
|
|
202
|
+
// change under this fix. A decimal seconds value like "0.17s" was matched
|
|
203
|
+
// by \b re-anchoring between the "." and the following digits, so "17s"
|
|
204
|
+
// alone got read as "seventeen seconds" and split the decimal apart; and
|
|
205
|
+
// the /i flag let a bare capital letter (money shorthand "5M", a memory
|
|
206
|
+
// size "4G", a product name "4080S") match a single-letter unit
|
|
207
|
+
// alternative it was never meant to.
|
|
208
|
+
it('decimal seconds survive intact — no mid-decimal unit match', () => {
|
|
209
|
+
expect(normalizeForSpeech('latency was 0.17s')).toBe('latency was 0.17s')
|
|
210
|
+
expect(normalizeForSpeech('took 1.5s to load')).toBe('took 1.5s to load')
|
|
211
|
+
expect(normalizeForSpeech('waited 2.25s')).toBe('waited 2.25s')
|
|
212
|
+
expect(normalizeForSpeech('under 0.9s')).toBe('under 0.9s')
|
|
213
|
+
})
|
|
214
|
+
|
|
215
|
+
// KNOWN GAP (disclosed, not fixed in this hotfix — see the comment above
|
|
216
|
+
// the unit-suffix regex in voice-normalize-text.ts): a decimal- or
|
|
217
|
+
// comma-glued number+unit token is now left completely unexpanded (unit
|
|
218
|
+
// unspoken) rather than mangled. Measured: samples carrying an unspoken
|
|
219
|
+
// digit-adjacent unit go 65 → 122 across the corpus (61 samples change,
|
|
220
|
+
// every one gaining an unspoken unit) — pinned here as the baseline a
|
|
221
|
+
// future decimal-expansion pass should move.
|
|
222
|
+
it('KNOWN GAP: decimal/comma-glued number+unit is left unexpanded, not mangled', () => {
|
|
223
|
+
expect(normalizeForSpeech('free: 13.0 GB')).toBe('free: 13.0 GB')
|
|
224
|
+
expect(normalizeForSpeech('weighs 89.5g')).toBe('weighs 89.5g')
|
|
225
|
+
expect(normalizeForSpeech('ran for 27.5 s')).toBe('ran for 27.5 s')
|
|
226
|
+
expect(normalizeForSpeech('paused 9.5 min')).toBe('paused 9.5 min')
|
|
227
|
+
})
|
|
228
|
+
|
|
229
|
+
// MINOR: comma leak — same defect class as the decimal lookbehind, closed
|
|
230
|
+
// in the same lookbehind by also excluding `,`. Before this: "1,500s" →
|
|
231
|
+
// "1,five hundred seconds", "12,000s" → "12,zero seconds", "1,024MB" →
|
|
232
|
+
// "1,twenty-four megabytes". The lookbehind is adjacency-only so an
|
|
233
|
+
// ordinary sentence comma ("wait, 5s") still expands correctly — only a
|
|
234
|
+
// digit glued directly to the comma is blocked.
|
|
235
|
+
it('comma-glued numbers are left unmangled; a sentence comma still expands', () => {
|
|
236
|
+
expect(normalizeForSpeech('rate limited: 1,500s')).toBe('rate limited: 1,500s')
|
|
237
|
+
expect(normalizeForSpeech('timeout after 12,000s')).toBe('timeout after 12,000s')
|
|
238
|
+
expect(normalizeForSpeech('buffer is 1,024MB')).toBe('buffer is 1,024MB')
|
|
239
|
+
expect(normalizeForSpeech('wait, 5s')).toBe('wait, five seconds')
|
|
240
|
+
})
|
|
241
|
+
|
|
242
|
+
it('whole-number seconds still expand: 90s → ninety seconds', () => {
|
|
243
|
+
expect(normalizeForSpeech('done in 90s')).toBe('done in ninety seconds')
|
|
244
|
+
})
|
|
245
|
+
|
|
246
|
+
it('units unaffected by the case-sensitivity fix: 500ms, 5MB, 2GB', () => {
|
|
247
|
+
expect(normalizeForSpeech('took 500ms')).toBe('took five hundred milliseconds')
|
|
248
|
+
expect(normalizeForSpeech('a 5MB image')).toBe('a five megabytes image')
|
|
249
|
+
expect(normalizeForSpeech('a 2GB dump')).toBe('a two gigabytes dump')
|
|
250
|
+
expect(normalizeForSpeech('run 5m')).toBe('run five minutes')
|
|
251
|
+
})
|
|
252
|
+
|
|
253
|
+
it('bare capital M never reads as minutes (money shorthand, corpus)', () => {
|
|
254
|
+
expect(normalizeForSpeech('Revenue was 5M this year')).toBe(
|
|
255
|
+
'Revenue was 5M this year',
|
|
256
|
+
)
|
|
257
|
+
// Pre-existing quirk, not introduced or fixed here: the currency regex
|
|
258
|
+
// consumes "$5" before the unit pass ever sees the dangling "M", so this
|
|
259
|
+
// never read as "minutes" even on unpatched main — pin the ACTUAL output
|
|
260
|
+
// instead of an assertion that can't fail in either direction.
|
|
261
|
+
expect(normalizeForSpeech('raised $5M in the round')).toBe(
|
|
262
|
+
'raised five dollarsM in the round',
|
|
263
|
+
)
|
|
264
|
+
})
|
|
265
|
+
|
|
266
|
+
it('HTTP-status-shaped input adjacent to a capital letter is not read as a duration', () => {
|
|
267
|
+
expect(normalizeForSpeech('a 503M error occurred')).toBe(
|
|
268
|
+
'a 503M error occurred',
|
|
269
|
+
)
|
|
270
|
+
expect(normalizeForSpeech('status 503M')).toBe('status 503M')
|
|
271
|
+
})
|
|
272
|
+
|
|
273
|
+
// Corpus wins from the case-sensitivity split (22 samples improved fleet-
|
|
274
|
+
// wide across both passes, 14 of them "G" memory sizes wrongly read via
|
|
275
|
+
// pass 2 as GRAMS, plus an RTX "4080S" wrongly read as a duration).
|
|
276
|
+
// Pass 1 has no "g" in its own UNIT_MAP, so these were already inert here
|
|
277
|
+
// pre-fix; pinned to lock in the no-op and guard the composed pipeline
|
|
278
|
+
// (see tts-normalize.test.ts for the pass-2 "grams"/"4080S" regression).
|
|
279
|
+
it('single-letter units case-split: bare capitals stay untouched', () => {
|
|
280
|
+
expect(normalizeForSpeech('done in 5S')).toBe('done in 5S')
|
|
281
|
+
expect(normalizeForSpeech('wait 3H')).toBe('wait 3H')
|
|
282
|
+
expect(normalizeForSpeech('ships in 10D')).toBe('ships in 10D')
|
|
283
|
+
expect(normalizeForSpeech('bought a 4080S')).toBe('bought a 4080S')
|
|
284
|
+
})
|
|
196
285
|
})
|
|
197
286
|
|
|
198
287
|
describe('normalizeForSpeech — abbreviations (phase 2)', () => {
|