@ai-matrx/media 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/voices.d.cts CHANGED
@@ -75,4 +75,119 @@ declare function voiceDisplayName(set: VoiceSetId, value: unknown): string;
75
75
  /** True when `id` is one of the live voice conversation (xAI Realtime) voices. */
76
76
  declare function isLiveConversationVoice(id: string): boolean;
77
77
 
78
- export { ASSISTANT_VOICE_ID, CARTESIA_API_VERSION, LEGACY_DEFAULT_VOICE_ID, LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, READING_VOICE_ID, TTS_DEFAULT_SPEED, TTS_DEFAULT_VOLUME, TTS_MODEL_ID, TTS_PLAYBACK_BUFFER_SEC, TTS_SPEED_MAX, TTS_SPEED_MIN, TTS_STREAMING_BUFFER_SEC, type VoiceOption, type VoicePurpose, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, resolveSpeed, resolveVoiceId, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };
78
+ /** A term→spoken-form substitution applied before any other speech parsing. */
79
+ interface SpeechPronunciation {
80
+ from: string;
81
+ to: string;
82
+ }
83
+ interface ParseMarkdownToTextOptions {
84
+ /**
85
+ * Custom Dictionary pronunciation substitutions. Applied FIRST — before the
86
+ * built-in abbreviation/number normalization — so a user term like "AI" can
87
+ * supply its own spoken form instead of the generic "A I" readout. Pairs
88
+ * should be sorted longest-first by the caller so multi-word terms replace
89
+ * before substrings.
90
+ */
91
+ pronunciations?: SpeechPronunciation[];
92
+ }
93
+ /**
94
+ * Long-form meanings retained as reference data for search, display, or future
95
+ * accessibility features. TTS must not use these expansions: repeatedly
96
+ * reading "Application Programming Interface" for API is not natural speech.
97
+ */
98
+ declare const COMMON_ABBREVIATION_EXPANSIONS: {
99
+ readonly AI: "Artificial Intelligence";
100
+ readonly API: "Application Programming Interface";
101
+ readonly HTTP: "Hypertext Transfer Protocol";
102
+ readonly HTTPS: "Hypertext Transfer Protocol Secure";
103
+ readonly URL: "Uniform Resource Locator";
104
+ readonly URI: "Uniform Resource Identifier";
105
+ readonly JSON: "JavaScript Object Notation";
106
+ readonly XML: "eXtensible Markup Language";
107
+ readonly CSS: "Cascading Style Sheets";
108
+ readonly HTML: "Hypertext Markup Language";
109
+ readonly JS: "JavaScript";
110
+ readonly TS: "TypeScript";
111
+ readonly SQL: "Structured Query Language";
112
+ readonly DB: "Database";
113
+ readonly UI: "User Interface";
114
+ readonly UX: "User Experience";
115
+ readonly SEO: "Search Engine Optimization";
116
+ readonly SDK: "Software Development Kit";
117
+ readonly CLI: "Command Line Interface";
118
+ readonly IDE: "Integrated Development Environment";
119
+ readonly JWT: "JSON Web Token";
120
+ readonly OAuth: "Open Authorization";
121
+ readonly REST: "Representational State Transfer";
122
+ readonly CRUD: "Create Read Update Delete";
123
+ readonly MVC: "Model View Controller";
124
+ readonly SPA: "Single Page Application";
125
+ readonly SSR: "Server Side Rendering";
126
+ readonly CSR: "Client Side Rendering";
127
+ readonly PWA: "Progressive Web App";
128
+ readonly DOM: "Document Object Model";
129
+ readonly BOM: "Browser Object Model";
130
+ readonly CDN: "Content Delivery Network";
131
+ readonly CMS: "Content Management System";
132
+ readonly ERP: "Enterprise Resource Planning";
133
+ readonly CRM: "Customer Relationship Management";
134
+ readonly SaaS: "Software as a Service";
135
+ readonly PaaS: "Platform as a Service";
136
+ readonly IaaS: "Infrastructure as a Service";
137
+ readonly VPN: "Virtual Private Network";
138
+ readonly LAN: "Local Area Network";
139
+ readonly WAN: "Wide Area Network";
140
+ readonly TCP: "Transmission Control Protocol";
141
+ readonly UDP: "User Datagram Protocol";
142
+ readonly IP: "Internet Protocol";
143
+ readonly DNS: "Domain Name System";
144
+ readonly FTP: "File Transfer Protocol";
145
+ readonly SMTP: "Simple Mail Transfer Protocol";
146
+ readonly POP3: "Post Office Protocol version 3";
147
+ readonly IMAP: "Internet Message Access Protocol";
148
+ };
149
+ /**
150
+ * Normalize technical abbreviations for both active TTS lanes.
151
+ *
152
+ * Cartesia Sonic 3.5 recommends single-space character delimiters for forced
153
+ * readout ("A P I"), and that plain-text form also works with the catalog's
154
+ * Groq/Orpheus engine, which has no SSML/pronunciation-dictionary input. Word
155
+ * acronyms stay conventional so models can say "json", "oh-auth", etc.
156
+ */
157
+ declare function normalizeSpeechAbbreviations(text: string): string;
158
+ /**
159
+ * The spoken word a fill-in-the-blank run of underscores becomes. One constant
160
+ * so every speech surface reads the same thing — change it here (e.g. "what")
161
+ * and TTS + generated flashcard fronts follow.
162
+ */
163
+ declare const SPEECH_BLANK_WORD = "blank";
164
+ /**
165
+ * Fill-in-the-blank underscores → a spoken word. A run of two-or-more
166
+ * underscores (any length — "___", "_____"), including the spaced style
167
+ * ("_ _ _"), reads as "blank" instead of the literal "underscore" every TTS
168
+ * engine would otherwise say. Generic and reusable: the centralized speech
169
+ * cleaner AND the generated flashcard spoken-fronts both route through this, so
170
+ * a blank sounds identical everywhere.
171
+ *
172
+ * Guards (so it only ever touches real blanks):
173
+ * - bounded by non-alphanumerics → never mangles `snake_case` or `__dunder__`;
174
+ * - a line of ONLY underscores is a markdown thematic break (divider), left
175
+ * untouched here for the horizontal-rule rule to drop.
176
+ */
177
+ declare function normalizeSpeechBlanks(text: string): string;
178
+ declare function parseMarkdownToText(markdown: string, options?: ParseMarkdownToTextOptions): string;
179
+
180
+ /**
181
+ * The languages read-aloud offers, with each language's own name — the union of the website's
182
+ * voice-language setting and the extension's quick picker (which added Persian).
183
+ */
184
+ interface ReadAloudLanguage {
185
+ code: string;
186
+ englishName: string;
187
+ nativeName: string;
188
+ }
189
+ declare const READ_ALOUD_LANGUAGES: readonly ReadAloudLanguage[];
190
+ /** The language for `code`, or English when the code is not offered. */
191
+ declare function readAloudLanguage(code: string | null | undefined): ReadAloudLanguage;
192
+
193
+ export { ASSISTANT_VOICE_ID, CARTESIA_API_VERSION, COMMON_ABBREVIATION_EXPANSIONS, LEGACY_DEFAULT_VOICE_ID, LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, type ParseMarkdownToTextOptions, READING_VOICE_ID, READ_ALOUD_LANGUAGES, type ReadAloudLanguage, SPEECH_BLANK_WORD, type SpeechPronunciation, TTS_DEFAULT_SPEED, TTS_DEFAULT_VOLUME, TTS_MODEL_ID, TTS_PLAYBACK_BUFFER_SEC, TTS_SPEED_MAX, TTS_SPEED_MIN, TTS_STREAMING_BUFFER_SEC, type VoiceOption, type VoicePurpose, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, normalizeSpeechAbbreviations, normalizeSpeechBlanks, parseMarkdownToText, readAloudLanguage, resolveSpeed, resolveVoiceId, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };
package/dist/voices.d.ts CHANGED
@@ -75,4 +75,119 @@ declare function voiceDisplayName(set: VoiceSetId, value: unknown): string;
75
75
  /** True when `id` is one of the live voice conversation (xAI Realtime) voices. */
76
76
  declare function isLiveConversationVoice(id: string): boolean;
77
77
 
78
- export { ASSISTANT_VOICE_ID, CARTESIA_API_VERSION, LEGACY_DEFAULT_VOICE_ID, LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, READING_VOICE_ID, TTS_DEFAULT_SPEED, TTS_DEFAULT_VOLUME, TTS_MODEL_ID, TTS_PLAYBACK_BUFFER_SEC, TTS_SPEED_MAX, TTS_SPEED_MIN, TTS_STREAMING_BUFFER_SEC, type VoiceOption, type VoicePurpose, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, resolveSpeed, resolveVoiceId, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };
78
+ /** A term→spoken-form substitution applied before any other speech parsing. */
79
+ interface SpeechPronunciation {
80
+ from: string;
81
+ to: string;
82
+ }
83
+ interface ParseMarkdownToTextOptions {
84
+ /**
85
+ * Custom Dictionary pronunciation substitutions. Applied FIRST — before the
86
+ * built-in abbreviation/number normalization — so a user term like "AI" can
87
+ * supply its own spoken form instead of the generic "A I" readout. Pairs
88
+ * should be sorted longest-first by the caller so multi-word terms replace
89
+ * before substrings.
90
+ */
91
+ pronunciations?: SpeechPronunciation[];
92
+ }
93
+ /**
94
+ * Long-form meanings retained as reference data for search, display, or future
95
+ * accessibility features. TTS must not use these expansions: repeatedly
96
+ * reading "Application Programming Interface" for API is not natural speech.
97
+ */
98
+ declare const COMMON_ABBREVIATION_EXPANSIONS: {
99
+ readonly AI: "Artificial Intelligence";
100
+ readonly API: "Application Programming Interface";
101
+ readonly HTTP: "Hypertext Transfer Protocol";
102
+ readonly HTTPS: "Hypertext Transfer Protocol Secure";
103
+ readonly URL: "Uniform Resource Locator";
104
+ readonly URI: "Uniform Resource Identifier";
105
+ readonly JSON: "JavaScript Object Notation";
106
+ readonly XML: "eXtensible Markup Language";
107
+ readonly CSS: "Cascading Style Sheets";
108
+ readonly HTML: "Hypertext Markup Language";
109
+ readonly JS: "JavaScript";
110
+ readonly TS: "TypeScript";
111
+ readonly SQL: "Structured Query Language";
112
+ readonly DB: "Database";
113
+ readonly UI: "User Interface";
114
+ readonly UX: "User Experience";
115
+ readonly SEO: "Search Engine Optimization";
116
+ readonly SDK: "Software Development Kit";
117
+ readonly CLI: "Command Line Interface";
118
+ readonly IDE: "Integrated Development Environment";
119
+ readonly JWT: "JSON Web Token";
120
+ readonly OAuth: "Open Authorization";
121
+ readonly REST: "Representational State Transfer";
122
+ readonly CRUD: "Create Read Update Delete";
123
+ readonly MVC: "Model View Controller";
124
+ readonly SPA: "Single Page Application";
125
+ readonly SSR: "Server Side Rendering";
126
+ readonly CSR: "Client Side Rendering";
127
+ readonly PWA: "Progressive Web App";
128
+ readonly DOM: "Document Object Model";
129
+ readonly BOM: "Browser Object Model";
130
+ readonly CDN: "Content Delivery Network";
131
+ readonly CMS: "Content Management System";
132
+ readonly ERP: "Enterprise Resource Planning";
133
+ readonly CRM: "Customer Relationship Management";
134
+ readonly SaaS: "Software as a Service";
135
+ readonly PaaS: "Platform as a Service";
136
+ readonly IaaS: "Infrastructure as a Service";
137
+ readonly VPN: "Virtual Private Network";
138
+ readonly LAN: "Local Area Network";
139
+ readonly WAN: "Wide Area Network";
140
+ readonly TCP: "Transmission Control Protocol";
141
+ readonly UDP: "User Datagram Protocol";
142
+ readonly IP: "Internet Protocol";
143
+ readonly DNS: "Domain Name System";
144
+ readonly FTP: "File Transfer Protocol";
145
+ readonly SMTP: "Simple Mail Transfer Protocol";
146
+ readonly POP3: "Post Office Protocol version 3";
147
+ readonly IMAP: "Internet Message Access Protocol";
148
+ };
149
+ /**
150
+ * Normalize technical abbreviations for both active TTS lanes.
151
+ *
152
+ * Cartesia Sonic 3.5 recommends single-space character delimiters for forced
153
+ * readout ("A P I"), and that plain-text form also works with the catalog's
154
+ * Groq/Orpheus engine, which has no SSML/pronunciation-dictionary input. Word
155
+ * acronyms stay conventional so models can say "json", "oh-auth", etc.
156
+ */
157
+ declare function normalizeSpeechAbbreviations(text: string): string;
158
+ /**
159
+ * The spoken word a fill-in-the-blank run of underscores becomes. One constant
160
+ * so every speech surface reads the same thing — change it here (e.g. "what")
161
+ * and TTS + generated flashcard fronts follow.
162
+ */
163
+ declare const SPEECH_BLANK_WORD = "blank";
164
+ /**
165
+ * Fill-in-the-blank underscores → a spoken word. A run of two-or-more
166
+ * underscores (any length — "___", "_____"), including the spaced style
167
+ * ("_ _ _"), reads as "blank" instead of the literal "underscore" every TTS
168
+ * engine would otherwise say. Generic and reusable: the centralized speech
169
+ * cleaner AND the generated flashcard spoken-fronts both route through this, so
170
+ * a blank sounds identical everywhere.
171
+ *
172
+ * Guards (so it only ever touches real blanks):
173
+ * - bounded by non-alphanumerics → never mangles `snake_case` or `__dunder__`;
174
+ * - a line of ONLY underscores is a markdown thematic break (divider), left
175
+ * untouched here for the horizontal-rule rule to drop.
176
+ */
177
+ declare function normalizeSpeechBlanks(text: string): string;
178
+ declare function parseMarkdownToText(markdown: string, options?: ParseMarkdownToTextOptions): string;
179
+
180
+ /**
181
+ * The languages read-aloud offers, with each language's own name — the union of the website's
182
+ * voice-language setting and the extension's quick picker (which added Persian).
183
+ */
184
+ interface ReadAloudLanguage {
185
+ code: string;
186
+ englishName: string;
187
+ nativeName: string;
188
+ }
189
+ declare const READ_ALOUD_LANGUAGES: readonly ReadAloudLanguage[];
190
+ /** The language for `code`, or English when the code is not offered. */
191
+ declare function readAloudLanguage(code: string | null | undefined): ReadAloudLanguage;
192
+
193
+ export { ASSISTANT_VOICE_ID, CARTESIA_API_VERSION, COMMON_ABBREVIATION_EXPANSIONS, LEGACY_DEFAULT_VOICE_ID, LIVE_CONVERSATION_SAMPLE_MODEL, LIVE_CONVERSATION_VOICES, type ParseMarkdownToTextOptions, READING_VOICE_ID, READ_ALOUD_LANGUAGES, type ReadAloudLanguage, SPEECH_BLANK_WORD, type SpeechPronunciation, TTS_DEFAULT_SPEED, TTS_DEFAULT_VOLUME, TTS_MODEL_ID, TTS_PLAYBACK_BUFFER_SEC, TTS_SPEED_MAX, TTS_SPEED_MIN, TTS_STREAMING_BUFFER_SEC, type VoiceOption, type VoicePurpose, type VoiceSetId, allVoices, availableVoices, isLiveConversationVoice, normalizeSpeechAbbreviations, normalizeSpeechBlanks, parseMarkdownToText, readAloudLanguage, resolveSpeed, resolveVoiceId, voiceDisplayName, voiceOptions, voiceSetDefaultLabel, voiceSetOf };