switchroom 0.21.9 → 0.21.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -350,16 +350,54 @@ export function normalizeForTts(text: string): string {
350
350
 
351
351
  // -- Number + unit suffix glued to the number (500ms, 5km, 10GB). Only a
352
352
  // known unit bounded by a non-letter, so identifiers never match.
353
+ //
354
+ // The leading `(?<![\d.,])` lookbehind is load-bearing: without it
355
+ // `\b` is satisfied between a decimal point (or thousands comma) and
356
+ // the digits that follow, so "17s" inside "0.17s" — or "500s" inside
357
+ // "1,500s" — would match on its own and mangle the number ("0.17s" →
358
+ // "0.seventeen seconds"; "1,500s" → "1,five hundred seconds"). Excludes
359
+ // both `.` and `,` so a glued decimal/thousands-separated number is
360
+ // never re-anchored on. Multi-letter units (ms, GB, …) stay
361
+ // case-insensitive since `5MB`/`500ms` must keep working regardless of
362
+ // case; single-letter units (s, m, h, d, g) are matched
363
+ // case-sensitively lowercase-only in a second pass — the single letter
364
+ // "M" is one keystroke from meaning "million" in prose ("$5M",
365
+ // "Revenue was 5M") and must never fall through to "minutes" just
366
+ // because the whole alternation ran under /i.
367
+ //
368
+ // KNOWN GAP, disclosed deliberately (2026-08, hotfix for #2760-derived
369
+ // corpus regression): a decimal- or comma-glued number with a unit
370
+ // (27.5s, 1,500s, 89.5g, 0.7-0.8m) is now left completely UNEXPANDED —
371
+ // the unit goes unspoken — rather than mangled the way it was before.
372
+ // Corpus-measured: samples carrying a digit-adjacent unspoken unit go
373
+ // 65 → 122 (61 samples change under this fix, every one gaining an
374
+ // unspoken unit). Correctly expanding a decimal/comma-glued number is
375
+ // explicitly out of scope for this narrow hotfix — see the pinning
376
+ // tests below (13.0 GB / 89.5g / 27.5 s) marking the baseline for a
377
+ // follow-up decimal-expansion pass.
378
+ //
379
+ // ASYMMETRY WITH pass 1 (voice-normalize-text.ts), pre-existing and
380
+ // NOT introduced by this fix: this pass allows an optional space
381
+ // (`\s?`) between the number and the unit and maps `m` to METRE; pass
382
+ // 1 has no `\s?` and maps `m` to MINUTE. That accident is exactly what
383
+ // makes "5m" (no space) come out of pass 1 as minutes and "16 m"
384
+ // (space) come out of pass 2 as metres in the composed pipeline —
385
+ // pinned by the composed-pipeline test in the test suite.
386
+ const unitReplacer = (m: string, num: string, unitRaw: string): string => {
387
+ const unit = UNIT_MAP[unitRaw.toLowerCase()]
388
+ if (!unit) return m
389
+ const n = Number(num)
390
+ const w = numberToWords(n)
391
+ if (!w) return m
392
+ return `${w} ${n === 1 ? unit.s : unit.p}`
393
+ }
353
394
  s = s.replace(
354
- /\b(\d{1,9})\s?(ms|sec|min|km|cm|mm|kg|kb|mb|gb|tb|ghz|mhz|hr|mi|s|m|h|d|g)(?![a-z])/gi,
355
- (m, num: string, unitRaw: string) => {
356
- const unit = UNIT_MAP[unitRaw.toLowerCase()]
357
- if (!unit) return m
358
- const n = Number(num)
359
- const w = numberToWords(n)
360
- if (!w) return m
361
- return `${w} ${n === 1 ? unit.s : unit.p}`
362
- },
395
+ /(?<![\d.,])\b(\d{1,9})\s?(ms|sec|min|km|cm|mm|kg|kb|mb|gb|tb|ghz|mhz|hr|mi)(?![a-zA-Z])/gi,
396
+ unitReplacer,
397
+ )
398
+ s = s.replace(
399
+ /(?<![\d.,])\b(\d{1,9})\s?(s|m|h|d|g)(?![a-zA-Z])/g,
400
+ unitReplacer,
363
401
  )
364
402
 
365
403
  // -- Symbols in prose.
@@ -561,16 +561,55 @@ export function normalizeForSpeech(input: string): string {
561
561
  // Number + unit suffix → "<number> <unit>" (e.g. 500ms, 2h, 10KB).
562
562
  // Only when the suffix is a known unit glued directly to the number and
563
563
  // bounded by a non-letter (so "my5thing" / "class5" are never touched).
564
+ //
565
+ // The leading `(?<![\d.,])` lookbehind is load-bearing: without it
566
+ // `\b` is satisfied between a decimal point (or thousands comma) and
567
+ // the digits that follow, so "17s" inside "0.17s" — or "500s" inside
568
+ // "1,500s" — would match on its own and mangle the number ("0.17s" →
569
+ // "0.seventeen seconds"; "1,500s" → "1,five hundred seconds"). Excludes
570
+ // both `.` and `,` so a glued decimal/thousands-separated number is
571
+ // never re-anchored on. Multi-letter units (ms, GB, …) stay
572
+ // case-insensitive since `5MB`/`500ms` must keep working regardless of
573
+ // case; single-letter units (s, m, h, d) are matched case-sensitively
574
+ // lowercase-only in a second pass — the single letter "M" is one
575
+ // keystroke from meaning "million" in prose ("$5M", "Revenue was 5M")
576
+ // and must never fall through to "minutes" just because the whole
577
+ // alternation ran under /i.
578
+ //
579
+ // KNOWN GAP, disclosed deliberately (2026-08, hotfix for a corpus
580
+ // regression): a decimal- or comma-glued number with a unit (27.5s,
581
+ // 1,500s, 89.5g, 0.7-0.8m) is now left completely UNEXPANDED — the
582
+ // unit goes unspoken — rather than mangled the way it was before.
583
+ // Corpus-measured: samples carrying a digit-adjacent unspoken unit go
584
+ // 65 → 122 (61 samples change under this fix, every one gaining an
585
+ // unspoken unit). Correctly expanding a decimal/comma-glued number is
586
+ // explicitly out of scope for this narrow hotfix — see the pinning
587
+ // tests below (13.0 GB / 89.5g / 27.5 s) marking the baseline for a
588
+ // follow-up decimal-expansion pass.
589
+ //
590
+ // ASYMMETRY WITH pass 2 (tts-normalize.ts), pre-existing and NOT
591
+ // introduced by this fix: this pass has NO optional space between the
592
+ // number and the unit (unlike pass 2's `\s?`) and maps `m` to MINUTE;
593
+ // pass 2 maps `m` to METRE. That accident is exactly what makes "5m"
594
+ // (no space) come out of THIS pass as minutes and "16 m" (space) fall
595
+ // through untouched here but come out of pass 2 as metres in the
596
+ // composed pipeline — pinned by the composed-pipeline test in the
597
+ // tts-normalize test suite.
598
+ const unitReplacer = (m: string, num: string, unitRaw: string): string => {
599
+ const unit = UNIT_MAP[unitRaw.toLowerCase()]
600
+ if (!unit) return m
601
+ const n = Number(num)
602
+ const w = numberToWords(n)
603
+ if (!w) return m
604
+ return `${w} ${n === 1 ? unit.s : unit.p}`
605
+ }
606
+ s = s.replace(
607
+ /(?<![\d.,])\b(\d{1,9})(ms|sec|min|kb|mb|gb|tb|hr)(?![a-zA-Z])/gi,
608
+ unitReplacer,
609
+ )
564
610
  s = s.replace(
565
- /\b(\d{1,9})(ms|sec|min|kb|mb|gb|tb|hr|s|m|h|d)(?![a-z])/gi,
566
- (m, num, unitRaw) => {
567
- const unit = UNIT_MAP[unitRaw.toLowerCase()]
568
- if (!unit) return m
569
- const n = Number(num)
570
- const w = numberToWords(n)
571
- if (!w) return m
572
- return `${w} ${n === 1 ? unit.s : unit.p}`
573
- },
611
+ /(?<![\d.,])\b(\d{1,9})(s|m|h|d)(?![a-zA-Z])/g,
612
+ unitReplacer,
574
613
  )
575
614
  // Symbols between tokens: standalone & → and, + → plus, = → equals.
576
615
  s = s.replace(/(\S)\s*\+\s*(\S)/g, '$1 plus $2')