@ai-matrx/media 0.13.0 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -0
- package/dist/react.cjs +201 -1
- package/dist/react.cjs.map +1 -1
- package/dist/react.d.cts +59 -2
- package/dist/react.d.ts +59 -2
- package/dist/react.js +210 -1
- package/dist/react.js.map +1 -1
- package/dist/speech.cjs +1689 -0
- package/dist/speech.cjs.map +1 -0
- package/dist/speech.d.cts +565 -0
- package/dist/speech.d.ts +565 -0
- package/dist/speech.js +1683 -0
- package/dist/speech.js.map +1 -0
- package/dist/tts-config-JbF86KIt.d.cts +149 -0
- package/dist/tts-config-JbF86KIt.d.ts +149 -0
- package/dist/voices.cjs +481 -0
- package/dist/voices.cjs.map +1 -1
- package/dist/voices.d.cts +16 -47
- package/dist/voices.d.ts +16 -47
- package/dist/voices.js +489 -0
- package/dist/voices.js.map +1 -1
- package/package.json +17 -4
package/dist/voices.d.cts
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
export { A as ASSISTANT_VOICE_ID, C as CARTESIA_API_VERSION, a as COMMON_ABBREVIATION_EXPANSIONS, L as LEGACY_DEFAULT_VOICE_ID, P as ParseMarkdownToTextOptions, R as READING_VOICE_ID, S as SPEECH_BLANK_WORD, b as SpeechPronunciation, T as TTS_DEFAULT_SPEED, c as TTS_DEFAULT_VOLUME, d as TTS_MODEL_ID, e as TTS_PLAYBACK_BUFFER_SEC, f as TTS_SPEED_MAX, g as TTS_SPEED_MIN, h as TTS_STREAMING_BUFFER_SEC, V as VoicePurpose, n as normalizeSpeechAbbreviations, i as normalizeSpeechBlanks, p as parseMarkdownToText, r as resolveSpeed, j as resolveVoiceId } from './tts-config-JbF86KIt.cjs';
|
|
2
|
+
|
|
1
3
|
declare const availableVoices: {
|
|
2
4
|
id: string;
|
|
3
5
|
name: string;
|
|
@@ -10,52 +12,6 @@ declare const allVoices: {
|
|
|
10
12
|
description: string;
|
|
11
13
|
}[];
|
|
12
14
|
|
|
13
|
-
/**
|
|
14
|
-
* Central Cartesia TTS configuration — the single source of truth for the model,
|
|
15
|
-
* API version, system default voices, speed/volume baselines, and playback
|
|
16
|
-
* buffering used by every in-app TTS surface (chat read-aloud, studio
|
|
17
|
-
* read-aloud, voice playgrounds, the admin tester).
|
|
18
|
-
*
|
|
19
|
-
* Moved from matrx-frontend lib/cartesia/config.ts (P16 voices). The SDK-typed
|
|
20
|
-
* `buildGenerationConfig` stays with the host's Cartesia client.
|
|
21
|
-
*/
|
|
22
|
-
/** Current model + API version for all in-app TTS. */
|
|
23
|
-
declare const TTS_MODEL_ID = "sonic-3.5";
|
|
24
|
-
declare const CARTESIA_API_VERSION = "2026-08-14";
|
|
25
|
-
/**
|
|
26
|
-
* System default voices, used only when a user has not chosen their own.
|
|
27
|
-
* - reading → Skylar (primary female; document / read-aloud)
|
|
28
|
-
* - assistant → Daniel (primary male; assistant replies)
|
|
29
|
-
*/
|
|
30
|
-
declare const READING_VOICE_ID = "db6b0ed5-d5d3-463d-ae85-518a07d3c2b4";
|
|
31
|
-
declare const ASSISTANT_VOICE_ID = "47c38ca4-5f35-497b-b1a3-415245fb35e1";
|
|
32
|
-
/**
|
|
33
|
-
* The pre-2026 hardcoded default voice. Treated as "unset" by resolveVoiceId so
|
|
34
|
-
* users who never explicitly chose a voice transition to the new defaults.
|
|
35
|
-
*/
|
|
36
|
-
declare const LEGACY_DEFAULT_VOICE_ID = "156fb8d2-335b-4950-9cb3-a2d33befec77";
|
|
37
|
-
type VoicePurpose = "reading" | "assistant";
|
|
38
|
-
/** generation_config.speed range (1.0 = original). Our chosen baseline is 1.2. */
|
|
39
|
-
declare const TTS_SPEED_MIN = 0.6;
|
|
40
|
-
declare const TTS_SPEED_MAX = 1.5;
|
|
41
|
-
declare const TTS_DEFAULT_SPEED = 1.2;
|
|
42
|
-
/** generation_config.volume range (1.0 = original). */
|
|
43
|
-
declare const TTS_DEFAULT_VOLUME = 1;
|
|
44
|
-
/**
|
|
45
|
-
* Client-side player buffer (seconds). Higher than the old 0.25s, which
|
|
46
|
-
* caused stream underruns heard as choppy "pauses"; tune in one place.
|
|
47
|
-
*/
|
|
48
|
-
declare const TTS_PLAYBACK_BUFFER_SEC = 0.7;
|
|
49
|
-
/**
|
|
50
|
-
* Lower buffer for token-by-token streaming (real-time LLM speech) where
|
|
51
|
-
* latency matters more — still well above the old 0.1s that stuttered.
|
|
52
|
-
*/
|
|
53
|
-
declare const TTS_STREAMING_BUFFER_SEC = 0.3;
|
|
54
|
-
/** A user's explicit voice preference wins; otherwise the purpose default. */
|
|
55
|
-
declare function resolveVoiceId(userVoiceId: string | null | undefined, purpose: VoicePurpose): string;
|
|
56
|
-
/** Clamp a stored speed to the valid generation_config range, else the default. */
|
|
57
|
-
declare function resolveSpeed(userSpeed: number | null | undefined): number;
|
|
58
|
-
|
|
59
15
|
type VoiceSetId = "cartesia" | "xai";
|
|
60
16
|
interface VoiceOption {
|
|
61
17
|
id: string;
|
|
@@ -75,4 +31,17 @@ declare function voiceDisplayName(set: VoiceSetId, value: unknown): string;
|
|
|
75
31
|
/** True when `id` is one of the live voice conversation (xAI Realtime) voices. */
|
|
76
32
|
declare function isLiveConversationVoice(id: string): boolean;
|
|
77
33
|
|
|
78
|
-
|
|
34
|
+
/**
|
|
35
|
+
* The languages read-aloud offers, with each language's own name — the union of the website's
|
|
36
|
+
* voice-language setting and the extension's quick picker (which added Persian).
|
|
37
|
+
*/
|
|
38
|
+
interface ReadAloudLanguage {
|
|
39
|
+
code: string;
|
|
40
|
+
englishName: string;
|
|
41
|
+
nativeName: string;
|
|
42
|
+
}
|
|
43
|
+
declare const READ_ALOUD_LANGUAGES: readonly ReadAloudLanguage[];
|
|
44
|
+
/** The language for `code`, or English when the code is not offered. */
|
|
45
|
+
declare function readAloudLanguage(code: string | null | undefined): ReadAloudLanguage;
|
|
46
|
+
|
|
47
|
+
export { LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, READ_ALOUD_LANGUAGES, type ReadAloudLanguage, type VoiceOption, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, readAloudLanguage, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };
|
package/dist/voices.d.ts
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
export { A as ASSISTANT_VOICE_ID, C as CARTESIA_API_VERSION, a as COMMON_ABBREVIATION_EXPANSIONS, L as LEGACY_DEFAULT_VOICE_ID, P as ParseMarkdownToTextOptions, R as READING_VOICE_ID, S as SPEECH_BLANK_WORD, b as SpeechPronunciation, T as TTS_DEFAULT_SPEED, c as TTS_DEFAULT_VOLUME, d as TTS_MODEL_ID, e as TTS_PLAYBACK_BUFFER_SEC, f as TTS_SPEED_MAX, g as TTS_SPEED_MIN, h as TTS_STREAMING_BUFFER_SEC, V as VoicePurpose, n as normalizeSpeechAbbreviations, i as normalizeSpeechBlanks, p as parseMarkdownToText, r as resolveSpeed, j as resolveVoiceId } from './tts-config-JbF86KIt.js';
|
|
2
|
+
|
|
1
3
|
declare const availableVoices: {
|
|
2
4
|
id: string;
|
|
3
5
|
name: string;
|
|
@@ -10,52 +12,6 @@ declare const allVoices: {
|
|
|
10
12
|
description: string;
|
|
11
13
|
}[];
|
|
12
14
|
|
|
13
|
-
/**
|
|
14
|
-
* Central Cartesia TTS configuration — the single source of truth for the model,
|
|
15
|
-
* API version, system default voices, speed/volume baselines, and playback
|
|
16
|
-
* buffering used by every in-app TTS surface (chat read-aloud, studio
|
|
17
|
-
* read-aloud, voice playgrounds, the admin tester).
|
|
18
|
-
*
|
|
19
|
-
* Moved from matrx-frontend lib/cartesia/config.ts (P16 voices). The SDK-typed
|
|
20
|
-
* `buildGenerationConfig` stays with the host's Cartesia client.
|
|
21
|
-
*/
|
|
22
|
-
/** Current model + API version for all in-app TTS. */
|
|
23
|
-
declare const TTS_MODEL_ID = "sonic-3.5";
|
|
24
|
-
declare const CARTESIA_API_VERSION = "2026-08-14";
|
|
25
|
-
/**
|
|
26
|
-
* System default voices, used only when a user has not chosen their own.
|
|
27
|
-
* - reading → Skylar (primary female; document / read-aloud)
|
|
28
|
-
* - assistant → Daniel (primary male; assistant replies)
|
|
29
|
-
*/
|
|
30
|
-
declare const READING_VOICE_ID = "db6b0ed5-d5d3-463d-ae85-518a07d3c2b4";
|
|
31
|
-
declare const ASSISTANT_VOICE_ID = "47c38ca4-5f35-497b-b1a3-415245fb35e1";
|
|
32
|
-
/**
|
|
33
|
-
* The pre-2026 hardcoded default voice. Treated as "unset" by resolveVoiceId so
|
|
34
|
-
* users who never explicitly chose a voice transition to the new defaults.
|
|
35
|
-
*/
|
|
36
|
-
declare const LEGACY_DEFAULT_VOICE_ID = "156fb8d2-335b-4950-9cb3-a2d33befec77";
|
|
37
|
-
type VoicePurpose = "reading" | "assistant";
|
|
38
|
-
/** generation_config.speed range (1.0 = original). Our chosen baseline is 1.2. */
|
|
39
|
-
declare const TTS_SPEED_MIN = 0.6;
|
|
40
|
-
declare const TTS_SPEED_MAX = 1.5;
|
|
41
|
-
declare const TTS_DEFAULT_SPEED = 1.2;
|
|
42
|
-
/** generation_config.volume range (1.0 = original). */
|
|
43
|
-
declare const TTS_DEFAULT_VOLUME = 1;
|
|
44
|
-
/**
|
|
45
|
-
* Client-side player buffer (seconds). Higher than the old 0.25s, which
|
|
46
|
-
* caused stream underruns heard as choppy "pauses"; tune in one place.
|
|
47
|
-
*/
|
|
48
|
-
declare const TTS_PLAYBACK_BUFFER_SEC = 0.7;
|
|
49
|
-
/**
|
|
50
|
-
* Lower buffer for token-by-token streaming (real-time LLM speech) where
|
|
51
|
-
* latency matters more — still well above the old 0.1s that stuttered.
|
|
52
|
-
*/
|
|
53
|
-
declare const TTS_STREAMING_BUFFER_SEC = 0.3;
|
|
54
|
-
/** A user's explicit voice preference wins; otherwise the purpose default. */
|
|
55
|
-
declare function resolveVoiceId(userVoiceId: string | null | undefined, purpose: VoicePurpose): string;
|
|
56
|
-
/** Clamp a stored speed to the valid generation_config range, else the default. */
|
|
57
|
-
declare function resolveSpeed(userSpeed: number | null | undefined): number;
|
|
58
|
-
|
|
59
15
|
type VoiceSetId = "cartesia" | "xai";
|
|
60
16
|
interface VoiceOption {
|
|
61
17
|
id: string;
|
|
@@ -75,4 +31,17 @@ declare function voiceDisplayName(set: VoiceSetId, value: unknown): string;
|
|
|
75
31
|
/** True when `id` is one of the live voice conversation (xAI Realtime) voices. */
|
|
76
32
|
declare function isLiveConversationVoice(id: string): boolean;
|
|
77
33
|
|
|
78
|
-
|
|
34
|
+
/**
|
|
35
|
+
* The languages read-aloud offers, with each language's own name — the union of the website's
|
|
36
|
+
* voice-language setting and the extension's quick picker (which added Persian).
|
|
37
|
+
*/
|
|
38
|
+
interface ReadAloudLanguage {
|
|
39
|
+
code: string;
|
|
40
|
+
englishName: string;
|
|
41
|
+
nativeName: string;
|
|
42
|
+
}
|
|
43
|
+
declare const READ_ALOUD_LANGUAGES: readonly ReadAloudLanguage[];
|
|
44
|
+
/** The language for `code`, or English when the code is not offered. */
|
|
45
|
+
declare function readAloudLanguage(code: string | null | undefined): ReadAloudLanguage;
|
|
46
|
+
|
|
47
|
+
export { LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, READ_ALOUD_LANGUAGES, type ReadAloudLanguage, type VoiceOption, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, readAloudLanguage, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };
|
package/dist/voices.js
CHANGED
|
@@ -2314,13 +2314,498 @@ function voiceDisplayName(set, value) {
|
|
|
2314
2314
|
function isLiveConversationVoice(id) {
|
|
2315
2315
|
return LIVE_CONVERSATION_VOICES.some((voice) => voice.id === id);
|
|
2316
2316
|
}
|
|
2317
|
+
|
|
2318
|
+
// src/voices/speech-text.ts
|
|
2319
|
+
import {
|
|
2320
|
+
findTableEnd,
|
|
2321
|
+
isPipeLedRow,
|
|
2322
|
+
replaceFences,
|
|
2323
|
+
rowCells,
|
|
2324
|
+
tableStartsAt,
|
|
2325
|
+
unescapeCellPipes,
|
|
2326
|
+
unwrapCodeSpans
|
|
2327
|
+
} from "@ai-matrx/content-ir/source";
|
|
2328
|
+
function speakTables(text) {
|
|
2329
|
+
const lines = text.split("\n");
|
|
2330
|
+
const out = [];
|
|
2331
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
2332
|
+
if (!tableStartsAt(lines, i)) {
|
|
2333
|
+
out.push(lines[i] ?? "");
|
|
2334
|
+
continue;
|
|
2335
|
+
}
|
|
2336
|
+
const end = findTableEnd(lines, i);
|
|
2337
|
+
const headers = rowCells(lines[i] ?? "").map(unescapeCellPipes).filter((h) => h.length > 0);
|
|
2338
|
+
const rowCount = lines.slice(i + 2, end).filter((line) => line.trim().length > 1).length;
|
|
2339
|
+
const headerText = headers.length > 1 ? headers.slice(0, -1).join(", ") + ", and " + headers[headers.length - 1] : headers.length === 1 ? headers[0] : "unlabeled columns";
|
|
2340
|
+
const rowText = rowCount === 1 ? "one row" : rowCount > 0 ? `${rowCount} rows` : "no rows";
|
|
2341
|
+
out.push(`There is a table with ${rowText} of data provided for ${headerText}.`);
|
|
2342
|
+
i = end - 1;
|
|
2343
|
+
}
|
|
2344
|
+
return out.join("\n");
|
|
2345
|
+
}
|
|
2346
|
+
var SPOKEN_CODE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
2347
|
+
"javascript",
|
|
2348
|
+
"js",
|
|
2349
|
+
"typescript",
|
|
2350
|
+
"ts",
|
|
2351
|
+
"python",
|
|
2352
|
+
"py",
|
|
2353
|
+
"java",
|
|
2354
|
+
"csharp",
|
|
2355
|
+
"cs",
|
|
2356
|
+
"cpp",
|
|
2357
|
+
"c++",
|
|
2358
|
+
"c",
|
|
2359
|
+
"go",
|
|
2360
|
+
"rust",
|
|
2361
|
+
"php",
|
|
2362
|
+
"ruby",
|
|
2363
|
+
"swift",
|
|
2364
|
+
"kotlin",
|
|
2365
|
+
"scala",
|
|
2366
|
+
"sql",
|
|
2367
|
+
"bash",
|
|
2368
|
+
"shell",
|
|
2369
|
+
"powershell",
|
|
2370
|
+
"yaml",
|
|
2371
|
+
"yml",
|
|
2372
|
+
"json",
|
|
2373
|
+
"xml",
|
|
2374
|
+
"html",
|
|
2375
|
+
"css",
|
|
2376
|
+
"markdown",
|
|
2377
|
+
"md"
|
|
2378
|
+
]);
|
|
2379
|
+
var COMMON_ABBREVIATION_EXPANSIONS = {
|
|
2380
|
+
AI: "Artificial Intelligence",
|
|
2381
|
+
API: "Application Programming Interface",
|
|
2382
|
+
HTTP: "Hypertext Transfer Protocol",
|
|
2383
|
+
HTTPS: "Hypertext Transfer Protocol Secure",
|
|
2384
|
+
URL: "Uniform Resource Locator",
|
|
2385
|
+
URI: "Uniform Resource Identifier",
|
|
2386
|
+
JSON: "JavaScript Object Notation",
|
|
2387
|
+
XML: "eXtensible Markup Language",
|
|
2388
|
+
CSS: "Cascading Style Sheets",
|
|
2389
|
+
HTML: "Hypertext Markup Language",
|
|
2390
|
+
JS: "JavaScript",
|
|
2391
|
+
TS: "TypeScript",
|
|
2392
|
+
SQL: "Structured Query Language",
|
|
2393
|
+
DB: "Database",
|
|
2394
|
+
UI: "User Interface",
|
|
2395
|
+
UX: "User Experience",
|
|
2396
|
+
SEO: "Search Engine Optimization",
|
|
2397
|
+
SDK: "Software Development Kit",
|
|
2398
|
+
CLI: "Command Line Interface",
|
|
2399
|
+
IDE: "Integrated Development Environment",
|
|
2400
|
+
JWT: "JSON Web Token",
|
|
2401
|
+
OAuth: "Open Authorization",
|
|
2402
|
+
REST: "Representational State Transfer",
|
|
2403
|
+
CRUD: "Create Read Update Delete",
|
|
2404
|
+
MVC: "Model View Controller",
|
|
2405
|
+
SPA: "Single Page Application",
|
|
2406
|
+
SSR: "Server Side Rendering",
|
|
2407
|
+
CSR: "Client Side Rendering",
|
|
2408
|
+
PWA: "Progressive Web App",
|
|
2409
|
+
DOM: "Document Object Model",
|
|
2410
|
+
BOM: "Browser Object Model",
|
|
2411
|
+
CDN: "Content Delivery Network",
|
|
2412
|
+
CMS: "Content Management System",
|
|
2413
|
+
ERP: "Enterprise Resource Planning",
|
|
2414
|
+
CRM: "Customer Relationship Management",
|
|
2415
|
+
SaaS: "Software as a Service",
|
|
2416
|
+
PaaS: "Platform as a Service",
|
|
2417
|
+
IaaS: "Infrastructure as a Service",
|
|
2418
|
+
VPN: "Virtual Private Network",
|
|
2419
|
+
LAN: "Local Area Network",
|
|
2420
|
+
WAN: "Wide Area Network",
|
|
2421
|
+
TCP: "Transmission Control Protocol",
|
|
2422
|
+
UDP: "User Datagram Protocol",
|
|
2423
|
+
IP: "Internet Protocol",
|
|
2424
|
+
DNS: "Domain Name System",
|
|
2425
|
+
FTP: "File Transfer Protocol",
|
|
2426
|
+
SMTP: "Simple Mail Transfer Protocol",
|
|
2427
|
+
POP3: "Post Office Protocol version 3",
|
|
2428
|
+
IMAP: "Internet Message Access Protocol"
|
|
2429
|
+
};
|
|
2430
|
+
function escapeRegExp(s) {
|
|
2431
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
2432
|
+
}
|
|
2433
|
+
var COMMON_ABBREVIATION_BY_UPPERCASE = Object.fromEntries(
|
|
2434
|
+
Object.keys(COMMON_ABBREVIATION_EXPANSIONS).map(
|
|
2435
|
+
(abbreviation) => [abbreviation.toUpperCase(), abbreviation]
|
|
2436
|
+
)
|
|
2437
|
+
);
|
|
2438
|
+
var COMMON_ABBREVIATION_PATTERN = new RegExp(
|
|
2439
|
+
`\\b(${Object.keys(COMMON_ABBREVIATION_EXPANSIONS).sort((a, b) => b.length - a.length).map(escapeRegExp).join("|")})(['\u2019]s|s)?\\b`,
|
|
2440
|
+
"gi"
|
|
2441
|
+
);
|
|
2442
|
+
var SPOKEN_AS_WORD = /* @__PURE__ */ new Set([
|
|
2443
|
+
"JSON",
|
|
2444
|
+
"OAuth",
|
|
2445
|
+
"REST",
|
|
2446
|
+
"CRUD",
|
|
2447
|
+
"DOM",
|
|
2448
|
+
"BOM",
|
|
2449
|
+
"SaaS",
|
|
2450
|
+
"PaaS",
|
|
2451
|
+
"IaaS",
|
|
2452
|
+
"LAN",
|
|
2453
|
+
"WAN"
|
|
2454
|
+
]);
|
|
2455
|
+
var CASE_SENSITIVE_WORD_ACRONYMS = /* @__PURE__ */ new Set([
|
|
2456
|
+
"REST",
|
|
2457
|
+
"CRUD",
|
|
2458
|
+
"SPA",
|
|
2459
|
+
"DOM",
|
|
2460
|
+
"BOM",
|
|
2461
|
+
"LAN",
|
|
2462
|
+
"WAN"
|
|
2463
|
+
]);
|
|
2464
|
+
function spokenFormForAbbreviation(match, sourceAbbreviation, suffix = "") {
|
|
2465
|
+
const abbreviation = COMMON_ABBREVIATION_BY_UPPERCASE[sourceAbbreviation.toUpperCase()];
|
|
2466
|
+
if (!abbreviation) return match;
|
|
2467
|
+
if (CASE_SENSITIVE_WORD_ACRONYMS.has(abbreviation) && sourceAbbreviation !== abbreviation) {
|
|
2468
|
+
return match;
|
|
2469
|
+
}
|
|
2470
|
+
if (SPOKEN_AS_WORD.has(abbreviation)) return `${abbreviation}${suffix}`;
|
|
2471
|
+
const spoken = Array.from(abbreviation.toUpperCase()).join(" ");
|
|
2472
|
+
return suffix ? `${spoken}'s` : spoken;
|
|
2473
|
+
}
|
|
2474
|
+
function normalizeSpeechAbbreviations(text) {
|
|
2475
|
+
if (!text) return text;
|
|
2476
|
+
return text.replace(COMMON_ABBREVIATION_PATTERN, spokenFormForAbbreviation);
|
|
2477
|
+
}
|
|
2478
|
+
var SPEECH_BLANK_WORD = "blank";
|
|
2479
|
+
function normalizeSpeechBlanks(text) {
|
|
2480
|
+
if (!text || !text.includes("_")) return text;
|
|
2481
|
+
const blank = /(?<![\p{L}\p{N}])_(?:[ \t]*_)+(?![\p{L}\p{N}])/gu;
|
|
2482
|
+
return text.split("\n").map(
|
|
2483
|
+
(line) => /^[ \t]*_[ \t_]*$/.test(line) ? line : line.replace(blank, ` ${SPEECH_BLANK_WORD} `)
|
|
2484
|
+
).join("\n");
|
|
2485
|
+
}
|
|
2486
|
+
function applyPronunciations(text, pairs) {
|
|
2487
|
+
let out = text;
|
|
2488
|
+
for (const { from, to } of pairs) {
|
|
2489
|
+
const term = from.trim();
|
|
2490
|
+
if (!term || !to) continue;
|
|
2491
|
+
const re = new RegExp(
|
|
2492
|
+
`(?<![\\p{L}\\p{N}])${escapeRegExp(term)}(?![\\p{L}\\p{N}])`,
|
|
2493
|
+
"giu"
|
|
2494
|
+
);
|
|
2495
|
+
out = out.replace(re, to);
|
|
2496
|
+
}
|
|
2497
|
+
return out;
|
|
2498
|
+
}
|
|
2499
|
+
function parseMarkdownToText(markdown, options) {
|
|
2500
|
+
if (!markdown || typeof markdown !== "string") return "";
|
|
2501
|
+
const numberToWords = (num) => {
|
|
2502
|
+
const ones = [
|
|
2503
|
+
"zero",
|
|
2504
|
+
"one",
|
|
2505
|
+
"two",
|
|
2506
|
+
"three",
|
|
2507
|
+
"four",
|
|
2508
|
+
"five",
|
|
2509
|
+
"six",
|
|
2510
|
+
"seven",
|
|
2511
|
+
"eight",
|
|
2512
|
+
"nine",
|
|
2513
|
+
"ten",
|
|
2514
|
+
"eleven",
|
|
2515
|
+
"twelve",
|
|
2516
|
+
"thirteen",
|
|
2517
|
+
"fourteen",
|
|
2518
|
+
"fifteen",
|
|
2519
|
+
"sixteen",
|
|
2520
|
+
"seventeen",
|
|
2521
|
+
"eighteen",
|
|
2522
|
+
"nineteen"
|
|
2523
|
+
];
|
|
2524
|
+
const tens = [
|
|
2525
|
+
"",
|
|
2526
|
+
"",
|
|
2527
|
+
"twenty",
|
|
2528
|
+
"thirty",
|
|
2529
|
+
"forty",
|
|
2530
|
+
"fifty",
|
|
2531
|
+
"sixty",
|
|
2532
|
+
"seventy",
|
|
2533
|
+
"eighty",
|
|
2534
|
+
"ninety"
|
|
2535
|
+
];
|
|
2536
|
+
const numInt = parseInt(num, 10);
|
|
2537
|
+
if (numInt < 20) return ones[numInt] ?? num;
|
|
2538
|
+
if (numInt < 100) {
|
|
2539
|
+
const t = Math.floor(numInt / 10);
|
|
2540
|
+
const o = numInt % 10;
|
|
2541
|
+
return o === 0 ? tens[t] ?? num : `${tens[t] ?? ""}-${ones[o] ?? ""}`;
|
|
2542
|
+
}
|
|
2543
|
+
if (numInt < 1e3) {
|
|
2544
|
+
const h = Math.floor(numInt / 100);
|
|
2545
|
+
const remainder = numInt % 100;
|
|
2546
|
+
return remainder === 0 ? `${ones[h]} hundred` : `${ones[h]} hundred ${numberToWords(String(remainder))}`;
|
|
2547
|
+
}
|
|
2548
|
+
return String(numInt);
|
|
2549
|
+
};
|
|
2550
|
+
const emojiMap = {
|
|
2551
|
+
"\u{1F60A}": "smiling face",
|
|
2552
|
+
"\u{1F602}": "laughing face",
|
|
2553
|
+
"\u2764\uFE0F": "heart",
|
|
2554
|
+
"\u{1F44D}": "thumbs up",
|
|
2555
|
+
"\u{1F44E}": "thumbs down",
|
|
2556
|
+
"\u{1F525}": "fire",
|
|
2557
|
+
"\u2B50": "star",
|
|
2558
|
+
"\u2705": "check mark",
|
|
2559
|
+
"\u274C": "cross mark",
|
|
2560
|
+
"\u26A0\uFE0F": "warning",
|
|
2561
|
+
"\u{1F680}": "rocket",
|
|
2562
|
+
"\u{1F4A1}": "light bulb",
|
|
2563
|
+
"\u{1F3AF}": "target",
|
|
2564
|
+
"\u{1F4F1}": "mobile phone",
|
|
2565
|
+
"\u{1F4BB}": "laptop",
|
|
2566
|
+
"\u{1F31F}": "glowing star",
|
|
2567
|
+
"\u{1F517}": "link",
|
|
2568
|
+
"\u{1F4E7}": "email",
|
|
2569
|
+
"\u{1F4C4}": "document",
|
|
2570
|
+
"\u{1F4CA}": "chart",
|
|
2571
|
+
"\u{1F6E0}\uFE0F": "tools",
|
|
2572
|
+
"\u2699\uFE0F": "gear",
|
|
2573
|
+
"\u{1F527}": "wrench",
|
|
2574
|
+
"\u{1F4C8}": "chart increasing",
|
|
2575
|
+
"\u{1F4C9}": "chart decreasing",
|
|
2576
|
+
"\u{1F389}": "party popper",
|
|
2577
|
+
"\u{1F4AA}": "flexed biceps",
|
|
2578
|
+
"\u{1F914}": "thinking face",
|
|
2579
|
+
"\u{1F4AD}": "thought bubble",
|
|
2580
|
+
"\u{1F440}": "eyes",
|
|
2581
|
+
"\u{1F44B}": "waving hand",
|
|
2582
|
+
"\u2728": "sparkles",
|
|
2583
|
+
"\u{1F50D}": "magnifying mx-glass",
|
|
2584
|
+
"\u{1F4DD}": "memo",
|
|
2585
|
+
"\u{1F4CB}": "clipboard",
|
|
2586
|
+
"\u{1F4C5}": "calendar",
|
|
2587
|
+
"\u23F0": "alarm clock",
|
|
2588
|
+
"\u{1F512}": "locked",
|
|
2589
|
+
"\u{1F513}": "unlocked",
|
|
2590
|
+
"\u{1F3C6}": "trophy",
|
|
2591
|
+
"\u{1F396}\uFE0F": "military medal",
|
|
2592
|
+
"\u{1F397}\uFE0F": "reminder ribbon",
|
|
2593
|
+
"\u{1F3C5}": "sports medal",
|
|
2594
|
+
"\u{1F947}": "first place medal",
|
|
2595
|
+
"\u{1F948}": "second place medal",
|
|
2596
|
+
"\u{1F949}": "third place medal"
|
|
2597
|
+
};
|
|
2598
|
+
const measurementUnits = {
|
|
2599
|
+
// Weight
|
|
2600
|
+
lbs: "pound",
|
|
2601
|
+
// Changed from 'pounds'
|
|
2602
|
+
lb: "pound",
|
|
2603
|
+
oz: "ounce",
|
|
2604
|
+
// Changed from 'ounces'
|
|
2605
|
+
kg: "kilogram",
|
|
2606
|
+
// Changed from 'kilograms'
|
|
2607
|
+
gm: "gram",
|
|
2608
|
+
// Changed from 'grams'
|
|
2609
|
+
mg: "milligram",
|
|
2610
|
+
// Changed from 'milligrams'
|
|
2611
|
+
// Distance/Length
|
|
2612
|
+
km: "kilometer",
|
|
2613
|
+
// Changed from 'kilometers'
|
|
2614
|
+
cm: "centimeter",
|
|
2615
|
+
// Changed from 'centimeters'
|
|
2616
|
+
mm: "millimeter",
|
|
2617
|
+
// Changed from 'millimeters'
|
|
2618
|
+
ft: "foot",
|
|
2619
|
+
// Changed from 'feet' (Irregular plural)
|
|
2620
|
+
mi: "mile",
|
|
2621
|
+
// Changed from 'miles'
|
|
2622
|
+
// Speed
|
|
2623
|
+
mph: "mile per hour",
|
|
2624
|
+
// Changed from 'miles per hour'
|
|
2625
|
+
kph: "kilometer per hour",
|
|
2626
|
+
// Changed from 'kilometers per hour'
|
|
2627
|
+
kmh: "kilometer per hour",
|
|
2628
|
+
// Changed from 'kilometers per hour'
|
|
2629
|
+
// Volume
|
|
2630
|
+
ml: "milliliter",
|
|
2631
|
+
// Changed from 'milliliters'
|
|
2632
|
+
mL: "milliliter",
|
|
2633
|
+
// Changed from 'milliliters'
|
|
2634
|
+
gal: "gallon",
|
|
2635
|
+
// Changed from 'gallons'
|
|
2636
|
+
qt: "quart",
|
|
2637
|
+
// Changed from 'quarts'
|
|
2638
|
+
tbsp: "tablespoon",
|
|
2639
|
+
// Changed from 'tablespoons'
|
|
2640
|
+
tsp: "teaspoon",
|
|
2641
|
+
// Changed from 'teaspoons'
|
|
2642
|
+
// Area
|
|
2643
|
+
sqft: "square foot",
|
|
2644
|
+
// Changed from 'square feet' (Irregular plural)
|
|
2645
|
+
sqm: "square meter",
|
|
2646
|
+
// Changed from 'square meters'
|
|
2647
|
+
// Time
|
|
2648
|
+
sec: "second",
|
|
2649
|
+
// Changed from 'seconds'
|
|
2650
|
+
min: "minute",
|
|
2651
|
+
// Changed from 'minutes'
|
|
2652
|
+
hr: "hour",
|
|
2653
|
+
// Changed from 'hours'
|
|
2654
|
+
hrs: "hour",
|
|
2655
|
+
// Changed from 'hours'
|
|
2656
|
+
ms: "millisecond",
|
|
2657
|
+
// Changed from 'milliseconds'
|
|
2658
|
+
// Other common units
|
|
2659
|
+
psi: "pound per square inch",
|
|
2660
|
+
// Changed from 'pounds per square inch'
|
|
2661
|
+
rpm: "revolution per minute",
|
|
2662
|
+
// Changed from 'revolutions per minute'
|
|
2663
|
+
bpm: "beat per minute"
|
|
2664
|
+
// Changed from 'beats per minute'
|
|
2665
|
+
};
|
|
2666
|
+
const stripReasoningTags = (input) => {
|
|
2667
|
+
return input.replace(/<(thinking|reasoning|think|reason)\b[^>]*>[\s\S]*?<\/\1>/gi, "").replace(/<\/?(?:thinking|reasoning|think|reason)\b[^>]*>/gi, "");
|
|
2668
|
+
};
|
|
2669
|
+
const stripLeadingMarkdown = (input) => {
|
|
2670
|
+
let text = input.replace(/^\uFEFF/, "").replace(/^[ \t]*\r?\n/, "");
|
|
2671
|
+
text = text.replace(
|
|
2672
|
+
/^[ \t]*(?:#{1,6}[ \t]+|>[ \t]*|[-*+][ \t]+|\d+\.[ \t]+|-[ \t]*\[[ xX]\][ \t]+)/,
|
|
2673
|
+
""
|
|
2674
|
+
);
|
|
2675
|
+
return text;
|
|
2676
|
+
};
|
|
2677
|
+
const source = normalizeSpeechBlanks(
|
|
2678
|
+
options?.pronunciations?.length ? applyPronunciations(markdown, options.pronunciations) : markdown
|
|
2679
|
+
);
|
|
2680
|
+
let result = speakTables(unwrapCodeSpans(replaceFences(stripLeadingMarkdown(stripReasoningTags(source)), ({ lang }) => {
|
|
2681
|
+
const language = lang.toLowerCase();
|
|
2682
|
+
if (language === "mermaid") return "Please see the diagram provided.";
|
|
2683
|
+
return SPOKEN_CODE_LANGUAGES.has(language) ? `Please see the ${language} code provided.` : "Please see the code provided.";
|
|
2684
|
+
}))).split("\n").map((line) => isPipeLedRow(line) && line.trimEnd().endsWith("|") ? "" : line).join("\n").replace(
|
|
2685
|
+
/\b(\d+)\s*[-–—]\s*(\d+)\b/g,
|
|
2686
|
+
(_match, a, b) => `${numberToWords(a)} to ${numberToWords(b)}`
|
|
2687
|
+
).replace(
|
|
2688
|
+
/^#{1,6}\s+(.+)$/gm,
|
|
2689
|
+
(_m, t) => /[.!?:]$/.test(t.trim()) ? t : `${t.trim()}.`
|
|
2690
|
+
).replace(/\/\/\s*/g, "").replace(/--\s*/g, "").replace(/#\s+/g, "").replace(/;\s*/g, "").replace(/(\*\*|__)(.*?)\1/g, "$2").replace(/(\*|_)(.*?)\1/g, "$2").replace(/~~(.*?)~~/g, "$1").replace(/==(.*?)==/g, "$1").replace(/\[([^\]]+)\]\(([^)]+)\)/g, (_match, text, url) => {
|
|
2691
|
+
if (url.startsWith("mailto:")) {
|
|
2692
|
+
return `${text}. Email address provided.`;
|
|
2693
|
+
}
|
|
2694
|
+
const fileExtensions = [
|
|
2695
|
+
".pdf",
|
|
2696
|
+
".doc",
|
|
2697
|
+
".docx",
|
|
2698
|
+
".xls",
|
|
2699
|
+
".xlsx",
|
|
2700
|
+
".ppt",
|
|
2701
|
+
".pptx",
|
|
2702
|
+
".txt",
|
|
2703
|
+
".md",
|
|
2704
|
+
".json",
|
|
2705
|
+
".xml",
|
|
2706
|
+
".csv",
|
|
2707
|
+
".zip"
|
|
2708
|
+
];
|
|
2709
|
+
const hasFileExtension = fileExtensions.some(
|
|
2710
|
+
(ext) => url.toLowerCase().includes(ext)
|
|
2711
|
+
);
|
|
2712
|
+
if (hasFileExtension) {
|
|
2713
|
+
return `${text}. Document link provided.`;
|
|
2714
|
+
}
|
|
2715
|
+
return `${text}. Link provided.`;
|
|
2716
|
+
}).replace(
|
|
2717
|
+
/!\[([^\]]*)\]\(([^)]+)\)/g,
|
|
2718
|
+
($0, alt) => alt ? `${alt}. Image provided.` : "Image provided."
|
|
2719
|
+
).replace(/^>\s*(.+)$/gm, "Quote: $1").replace(/^-\s*\[x\]\s+(.+)$/gm, "Completed task: $1").replace(/^-\s*\[\s*\]\s+(.+)$/gm, "Pending task: $1").replace(/^([-*+])\s+(.+)$/gm, "$2").replace(
|
|
2720
|
+
/^(\d+)\.\s+(.+)$/gm,
|
|
2721
|
+
(_match, num, content) => `Number ${numberToWords(num)}: ${content}`
|
|
2722
|
+
).replace(/\[\^(\d+)\]/g, "Reference $1").replace(/^\[\d+\]:\s*(.+)$/gm, "Reference: $1").replace(/^(\*{3,}|-{3,}|_{3,})$/gm, "").replace(/\$([^$]+)\$/g, "Mathematical expression: $1").replace(/\$\$([^$]+)\$\$/g, "Mathematical formula: $1").replace(
|
|
2723
|
+
/(\+\d{1,3}[-.\s]?)?\(?\d{3}\)?[-.\s]?\d{3}[-.\s]?\d{4}/g,
|
|
2724
|
+
"Phone number provided."
|
|
2725
|
+
).replace(
|
|
2726
|
+
/([a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,})/g,
|
|
2727
|
+
"Email address: $1"
|
|
2728
|
+
).replace(/\$(\d+(?:,\d{3})*(?:\.\d{2})?)/g, "$1 dollars").replace(/€(\d+(?:[.,]\d{3})*(?:[.,]\d{2})?)/g, "$1 euros").replace(/£(\d+(?:,\d{3})*(?:\.\d{2})?)/g, "$1 pounds").replace(/¥(\d+(?:,\d{3})*(?:\.\d{2})?)/g, "$1 yen").replace(
|
|
2729
|
+
/(\d{4})[-/](\d{1,2})[-/](\d{1,2})/g,
|
|
2730
|
+
(_match, year, month, day) => {
|
|
2731
|
+
const monthNames = [
|
|
2732
|
+
"January",
|
|
2733
|
+
"February",
|
|
2734
|
+
"March",
|
|
2735
|
+
"April",
|
|
2736
|
+
"May",
|
|
2737
|
+
"June",
|
|
2738
|
+
"July",
|
|
2739
|
+
"August",
|
|
2740
|
+
"September",
|
|
2741
|
+
"October",
|
|
2742
|
+
"November",
|
|
2743
|
+
"December"
|
|
2744
|
+
];
|
|
2745
|
+
const monthName = monthNames[parseInt(month) - 1] || month;
|
|
2746
|
+
return `${monthName} ${day}, ${year}`;
|
|
2747
|
+
}
|
|
2748
|
+
).replace(
|
|
2749
|
+
/(\d{1,2}):(\d{2})\s*(AM|PM|am|pm)?/g,
|
|
2750
|
+
(_match, hour, minute, period) => {
|
|
2751
|
+
const hourNum = parseInt(hour);
|
|
2752
|
+
const periodText = period ? period.toUpperCase() === "AM" ? "A.M." : "P.M." : "";
|
|
2753
|
+
return `${hourNum} ${minute} ${periodText}`.trim();
|
|
2754
|
+
}
|
|
2755
|
+
).replace(
|
|
2756
|
+
/[\u{1F600}-\u{1F64F}]|[\u{1F300}-\u{1F5FF}]|[\u{1F680}-\u{1F6FF}]|[\u{1F1E0}-\u{1F1FF}]|[\u{2600}-\u{26FF}]|[\u{2700}-\u{27BF}]/gu,
|
|
2757
|
+
(emoji) => {
|
|
2758
|
+
return emojiMap[emoji] || "emoji";
|
|
2759
|
+
}
|
|
2760
|
+
).replace(COMMON_ABBREVIATION_PATTERN, spokenFormForAbbreviation).replace(
|
|
2761
|
+
/(\d+(?:\.\d+)?)\s*(lbs|lb|oz|kg|gm|mg|km|cm|mm|ft|mi|mph|kph|kmh|ml|mL|gal|qt|tbsp|tsp|sqft|sqm|sec|min|hr|hrs|ms|psi|rpm|bpm)\b/gi,
|
|
2762
|
+
(match, number, unit) => {
|
|
2763
|
+
const unitLower = unit.toLowerCase();
|
|
2764
|
+
const unitText = measurementUnits[unitLower] || measurementUnits[unit] || unit;
|
|
2765
|
+
return `${number} ${unitText}`;
|
|
2766
|
+
}
|
|
2767
|
+
).replace(
|
|
2768
|
+
/\b(lbs|lb|oz|kg|gm|mg|km|cm|mm|mph|kph|kmh|ml|mL|gal|qt|tbsp|tsp|sqft|sqm|psi|rpm|bpm)(?=\s|$|[.,;!?])/gi,
|
|
2769
|
+
(match) => {
|
|
2770
|
+
const unitLower = match.toLowerCase();
|
|
2771
|
+
return measurementUnits[unitLower] || measurementUnits[match] || match;
|
|
2772
|
+
}
|
|
2773
|
+
).replace(/©/g, "copyright").replace(/®/g, "registered trademark").replace(/™/g, "trademark").replace(/°/g, "degrees").replace(/±/g, "plus or minus").replace(/≈/g, "approximately").replace(/≠/g, "not equal to").replace(/≤/g, "less than or equal to").replace(/≥/g, "greater than or equal to").replace(/=/g, " equals ").replace(/→/g, "arrow").replace(/←/g, "left arrow").replace(/↑/g, "up arrow").replace(/↓/g, "down arrow").replace(/<\/?[^>]+(>|$)/g, "").replace(/([^\s.!?,;:\-])(\n)/g, "$1.\n").replace(/\s+/g, " ").trim();
|
|
2774
|
+
return result;
|
|
2775
|
+
}
|
|
2776
|
+
|
|
2777
|
+
// src/voices/read-aloud-languages.ts
|
|
2778
|
+
var READ_ALOUD_LANGUAGES = [
|
|
2779
|
+
{ code: "en", englishName: "English", nativeName: "English" },
|
|
2780
|
+
{ code: "es", englishName: "Spanish", nativeName: "Espa\xF1ol" },
|
|
2781
|
+
{ code: "fr", englishName: "French", nativeName: "Fran\xE7ais" },
|
|
2782
|
+
{ code: "de", englishName: "German", nativeName: "Deutsch" },
|
|
2783
|
+
{ code: "it", englishName: "Italian", nativeName: "Italiano" },
|
|
2784
|
+
{ code: "pt", englishName: "Portuguese", nativeName: "Portugu\xEAs" },
|
|
2785
|
+
{ code: "zh", englishName: "Chinese", nativeName: "\u4E2D\u6587" },
|
|
2786
|
+
{ code: "ja", englishName: "Japanese", nativeName: "\u65E5\u672C\u8A9E" },
|
|
2787
|
+
{ code: "ko", englishName: "Korean", nativeName: "\uD55C\uAD6D\uC5B4" },
|
|
2788
|
+
{ code: "ru", englishName: "Russian", nativeName: "\u0420\u0443\u0441\u0441\u043A\u0438\u0439" },
|
|
2789
|
+
{ code: "hi", englishName: "Hindi", nativeName: "\u0939\u093F\u0928\u094D\u0926\u0940" },
|
|
2790
|
+
{ code: "nl", englishName: "Dutch", nativeName: "Nederlands" },
|
|
2791
|
+
{ code: "pl", englishName: "Polish", nativeName: "Polski" },
|
|
2792
|
+
{ code: "sv", englishName: "Swedish", nativeName: "Svenska" },
|
|
2793
|
+
{ code: "tr", englishName: "Turkish", nativeName: "T\xFCrk\xE7e" },
|
|
2794
|
+
{ code: "fa", englishName: "Persian", nativeName: "\u0641\u0627\u0631\u0633\u06CC" }
|
|
2795
|
+
];
|
|
2796
|
+
function readAloudLanguage(code) {
|
|
2797
|
+
return READ_ALOUD_LANGUAGES.find((l) => l.code === code) ?? READ_ALOUD_LANGUAGES[0];
|
|
2798
|
+
}
|
|
2317
2799
|
export {
|
|
2318
2800
|
ASSISTANT_VOICE_ID,
|
|
2319
2801
|
CARTESIA_API_VERSION,
|
|
2802
|
+
COMMON_ABBREVIATION_EXPANSIONS,
|
|
2320
2803
|
LEGACY_DEFAULT_VOICE_ID,
|
|
2321
2804
|
LIVE_CONVERSATION_SAMPLE_MODEL,
|
|
2322
2805
|
LIVE_CONVERSATION_VOICES,
|
|
2323
2806
|
READING_VOICE_ID,
|
|
2807
|
+
READ_ALOUD_LANGUAGES,
|
|
2808
|
+
SPEECH_BLANK_WORD,
|
|
2324
2809
|
TTS_DEFAULT_SPEED,
|
|
2325
2810
|
TTS_DEFAULT_VOLUME,
|
|
2326
2811
|
TTS_MODEL_ID,
|
|
@@ -2331,6 +2816,10 @@ export {
|
|
|
2331
2816
|
allVoices,
|
|
2332
2817
|
availableVoices,
|
|
2333
2818
|
isLiveConversationVoice,
|
|
2819
|
+
normalizeSpeechAbbreviations,
|
|
2820
|
+
normalizeSpeechBlanks,
|
|
2821
|
+
parseMarkdownToText,
|
|
2822
|
+
readAloudLanguage,
|
|
2334
2823
|
resolveSpeed,
|
|
2335
2824
|
resolveVoiceId,
|
|
2336
2825
|
voiceDisplayName,
|