@ai-matrx/media 0.13.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,565 @@
1
+ import { ProviderSessionFailure, ProviderSessionFailureVerdict } from '@ai-matrx/agents/matrx';
2
+ import { b as SpeechPronunciation, V as VoicePurpose } from './tts-config-JbF86KIt.cjs';
3
+ import { GenerationConfig } from '@cartesia/cartesia-js/resources/tts';
4
+ import { SupportedLanguage } from '@cartesia/cartesia-js/resources/voices';
5
+
6
+ /**
7
+ * Unified audio playback — shared types.
8
+ *
9
+ * One queue, one output, app-wide. Every "speak this" request becomes a
10
+ * `PlaybackItem` on the single `playbackQueue`. If audio is already playing, the
11
+ * new request is QUEUED (never overlapped — the playback twin of the capture
12
+ * lock's start-always-wins). Providers plug in via `PlaybackAdapter`s so the
13
+ * queue stays engine-agnostic (Cartesia WebSocket, Groq WAV blob, …).
14
+ */
15
+ /**
16
+ * Which synthesis engine renders an item. Declared with its capabilities in the
17
+ * AV engine registry (`features/audio/service/engines.ts`) — this union and
18
+ * `SpeakEngineId` are the same set by construction (see the type assertion
19
+ * there). Consumers call `speak()`, never these keys directly.
20
+ */
21
+ type PlaybackProvider = "cartesia" | "catalog" | (string & {});
22
+ type PlaybackItemStatus = "queued" | "loading" | "playing" | "paused" | "done" | "error";
23
+ /** What a consumer asks the queue to speak. */
24
+ interface PlaybackRequest {
25
+ provider: PlaybackProvider;
26
+ /** Raw text; the adapter applies markdown stripping per `processMarkdown`. */
27
+ text: string;
28
+ processMarkdown?: boolean | undefined;
29
+ /** Short human label for the queue UI (e.g. "Assistant reply", "Note"). */
30
+ label?: string | undefined;
31
+ /** Opt this utterance into Custom Dictionary pronunciation (Cartesia). */
32
+ dictionarySurfaceKey?: string | undefined;
33
+ /**
34
+ * Cartesia voice params — EXPLICIT overrides only. Anything absent is
35
+ * resolved by the adapter AT START TIME from the tiered listening config
36
+ * (`resolveListeningSettings`), so queued items and history replays honor the
37
+ * settings current when audio actually plays, not when it was enqueued.
38
+ * `purpose` picks the default-voice slot when no voice is set anywhere.
39
+ */
40
+ cartesia?: {
41
+ voiceId?: string | undefined;
42
+ language?: string | undefined;
43
+ speed?: number | undefined;
44
+ purpose?: "reading" | "assistant" | undefined;
45
+ };
46
+ /** Catalog speech voice params — resolved by `speak()`; server picks the vendor. */
47
+ catalog?: {
48
+ voice?: string | undefined;
49
+ model?: string | undefined;
50
+ /**
51
+ * Play a model's sample of one voice instead of synthesizing `text`: the
52
+ * vendor's own sample when it publishes one (ElevenLabs), otherwise one
53
+ * short line the server renders once and caches per (model, voice).
54
+ */
55
+ sample?: {
56
+ model: string;
57
+ voice: string;
58
+ offeringId?: string | undefined;
59
+ } | undefined;
60
+ };
61
+ }
62
+ interface PlaybackItem extends PlaybackRequest {
63
+ id: string;
64
+ status: PlaybackItemStatus;
65
+ error?: string | undefined;
66
+ enqueuedAtMs: number;
67
+ }
68
+ /** Serializable snapshot the queue pushes to subscribers (→ Redux mirror). */
69
+ interface PlaybackSnapshot {
70
+ items: PlaybackItem[];
71
+ /** Id of the item currently loading/playing/paused, or null when idle. */
72
+ currentId: string | null;
73
+ /** Global playback rate (applies live to rate-capable providers, e.g. Groq). */
74
+ rate: number;
75
+ }
76
+ /** Callbacks an adapter fires back into the queue as playback progresses. */
77
+ interface PlaybackAdapterCallbacks {
78
+ onLoading: () => void;
79
+ onPlaying: () => void;
80
+ /** Natural end — the queue advances to the next item. */
81
+ onEnded: () => void;
82
+ onError: (message: string) => void;
83
+ }
84
+ /** Handle for controlling the one active playback an adapter started. */
85
+ interface ActivePlayback {
86
+ pause: () => void | Promise<void>;
87
+ resume: () => void | Promise<void>;
88
+ /** Stop + release all resources. Must be idempotent. */
89
+ stop: () => void | Promise<void>;
90
+ /** Live playback-rate change (rate-capable providers only). */
91
+ setRate?: (rate: number) => void;
92
+ }
93
+ interface PlaybackAdapter {
94
+ provider: PlaybackProvider;
95
+ /**
96
+ * Begin playing `item`. Resolves with a control handle as soon as playback is
97
+ * underway (NOT when it finishes). Fire `cb.onPlaying` once audio starts and
98
+ * `cb.onEnded` when it finishes naturally.
99
+ */
100
+ start: (item: PlaybackItem, cb: PlaybackAdapterCallbacks, rate: number) => Promise<ActivePlayback>;
101
+ }
102
+
103
+ /** Voice parameters for one Cartesia utterance, resolved at START time. */
104
+ interface SpeechVoiceSettings {
105
+ /** The person's saved Cartesia voice id (null → the purpose default). */
106
+ voice: string | null;
107
+ language: string;
108
+ speed: number;
109
+ emotion?: string | null;
110
+ }
111
+ interface SpeechPorts {
112
+ /** A short-lived Cartesia access token. `forceRefresh` after the provider rejected the last one. */
113
+ getCartesiaToken: (opts?: {
114
+ forceRefresh?: boolean;
115
+ }) => Promise<string>;
116
+ /** Drop a token the provider rejected so the next read mints a fresh one. */
117
+ invalidateCartesiaToken?: (token: string) => void;
118
+ /** The person's voice / language / speed right now (tiered settings on the web). */
119
+ resolveVoiceSettings: () => Promise<SpeechVoiceSettings>;
120
+ /** Custom pronunciations for one utterance (best-effort; failures speak without them). */
121
+ resolvePronunciations?: (item: PlaybackItem) => Promise<SpeechPronunciation[]>;
122
+ /**
123
+ * Report a browser-held provider session failure to the server; the verdict carries the
124
+ * person's sentence and whether retrying can help.
125
+ */
126
+ reportProviderFailure?: (failure: ProviderSessionFailure) => Promise<ProviderSessionFailureVerdict | null>;
127
+ /** Tell the person speech stopped (a toast on the web). Without it the error is only on the item. */
128
+ notifyError?: (title: string, description: string) => void;
129
+ /**
130
+ * Runs before an item starts (e.g. "speaking happens in an organization" — hold until one is set).
131
+ * Throwing marks the item failed with `describeStartRefusal(err)` (or the error's message).
132
+ */
133
+ beforeStart?: (item: PlaybackItem) => Promise<void>;
134
+ describeStartRefusal?: (error: unknown) => string | null;
135
+ /** First engagement of the engine (the web mounts its lazy audio UI here). */
136
+ onEngage?: () => void;
137
+ /** Extra engines (the web's server catalog lane). Cartesia is built in. */
138
+ adapters?: Partial<Record<PlaybackProvider, () => Promise<PlaybackAdapter>>>;
139
+ /** Engine used when a `speak` call names none. Default "cartesia". */
140
+ defaultEngine?: PlaybackProvider;
141
+ /** Fill engine-specific params a `speak` call left out (the web's catalog voice preference). */
142
+ completeRequest?: (request: PlaybackRequest) => PlaybackRequest;
143
+ }
144
+ declare function configureSpeech(next: Partial<SpeechPorts>): void;
145
+ /** Test seam: forget every port. */
146
+ declare function resetSpeechPortsForTests(): void;
147
+
148
+ /**
149
+ * speak() — THE entry point for turning text into audio, in every app.
150
+ *
151
+ * One call, any engine. The caller says WHAT to say (and optionally which engine); the queue owns
152
+ * one-at-a-time playback, the app-wide playback lock and lazy-loading of the engine. Cartesia's
153
+ * voice / language / speed resolve at START time through the host's `resolveVoiceSettings` port,
154
+ * so queued items and replays honor the settings current when audio plays. Framework-free;
155
+ * React surfaces use `useSpeech()` (`@ai-matrx/media/react`) on top of the same call.
156
+ */
157
+
158
+ interface SpeakRequest {
159
+ text: string;
160
+ /** Engine to synthesize with. Omit for the host's default (Cartesia unless configured). */
161
+ engine?: PlaybackProvider | undefined;
162
+ /** Which voice slot to resolve for engines with per-purpose voices (Cartesia). */
163
+ purpose?: VoicePurpose | undefined;
164
+ /** Strip markdown before speaking (default true). */
165
+ processMarkdown?: boolean | undefined;
166
+ /** Short human label for a queue / Media panel row. */
167
+ label?: string | undefined;
168
+ /** Opt this utterance into Custom Dictionary pronunciation (host-resolved). */
169
+ dictionarySurfaceKey?: string | undefined;
170
+ /** Explicit voice id — overrides the saved voice for this utterance. */
171
+ voice?: string | undefined;
172
+ /** Explicit language — overrides the saved language. */
173
+ language?: string | undefined;
174
+ /** Explicit speed — overrides the saved speed. */
175
+ speed?: number | undefined;
176
+ /** Hear a catalog model's sample of one voice instead of speaking `text` (catalog engine). */
177
+ sample?: {
178
+ model: string;
179
+ voice: string;
180
+ offeringId?: string | undefined;
181
+ } | undefined;
182
+ }
183
+ interface SpeakResult {
184
+ /** Queue item id — feed it to the queue's pause/resume/remove verbs. */
185
+ id: string;
186
+ /** The engine that actually ran (after defaulting). */
187
+ engine: PlaybackProvider;
188
+ }
189
+ /** The engine for a request: explicit wins, otherwise the host default, otherwise Cartesia. */
190
+ declare function resolveSpeakEngine(requested?: PlaybackProvider): PlaybackProvider;
191
+ /** Speak text through the single playback queue. */
192
+ declare function speak(request: SpeakRequest): SpeakResult;
193
+
194
+ /**
195
+ * playbackQueue — the single, app-wide audio playback queue.
196
+ *
197
+ * Framework-free singleton (the playback twin of `captureLock`). Every "speak
198
+ * this" request becomes an item; audio plays ONE at a time. New requests while
199
+ * something is playing are appended (queued, never overlapped). Finished items
200
+ * stay in history (status `done`) until cleared so the UI can offer replay.
201
+ *
202
+ * Providers plug in via `PlaybackAdapter`s, so the queue never knows about
203
+ * Cartesia WebSockets or Groq blobs.
204
+ *
205
+ * Moved from matrx-frontend (features/audio/playback/playbackQueue.ts) into @ai-matrx/media so
206
+ * every app plays through ONE queue. A host may mirror snapshots (the web mirrors into Redux for
207
+ * its Media panel); React reads it through `useSpeech` / `usePlaybackSnapshot`.
208
+ */
209
+
210
+ type Listener = (snapshot: PlaybackSnapshot) => void;
211
+ interface EnqueueResult {
212
+ id: string;
213
+ }
214
+ /** Add a request to the queue. Plays immediately if nothing is active. */
215
+ declare function enqueuePlayback(request: PlaybackRequest): EnqueueResult;
216
+ declare function pausePlayback(): Promise<void>;
217
+ declare function resumePlayback(): Promise<void>;
218
+ /** Stop the active item (marks it done) and advance to the next queued item. */
219
+ declare function skipPlayback(): Promise<void>;
220
+ /** Play (or replay) a specific item now, taking over anything active. */
221
+ declare function playPlaybackItem(id: string): Promise<void>;
222
+ /** Remove an item. If it's the active one, advance. */
223
+ declare function removePlaybackItem(id: string): Promise<void>;
224
+ /** Stop everything and clear the queue + history. */
225
+ declare function clearPlayback(): Promise<void>;
226
+ declare function setPlaybackRate(value: number): void;
227
+ declare function getPlaybackSnapshot(): PlaybackSnapshot;
228
+ /** Test seam: drop every item and forget loaded engines. */
229
+ declare function resetPlaybackForTests(): void;
230
+ declare function subscribePlayback(listener: Listener): () => void;
231
+
232
+ /**
233
+ * Playback Lock — app-wide single-OUTPUT arbitration.
234
+ *
235
+ * The output twin of `captureLock`. Just as only one mic capture may be live at
236
+ * a time, **only one thing may be producing audible playback at any instant,
237
+ * anywhere in the app.** Multiple playback paths exist (the unified
238
+ * `playbackQueue` for Speaker buttons, the streaming auto-voice
239
+ * `useCartesiaStreamingSpeaker`, xAI realtime, podcast PCM) and each one used to
240
+ * play independently — so a War Room read-aloud could stream over a queued
241
+ * utterance, two voices in your ear at once. This arbiter makes that
242
+ * structurally impossible.
243
+ *
244
+ * Claiming is **start-always-wins**: a new claim instantly stops the current
245
+ * holder (its `stop` runs synchronously) and takes ownership. Framework-free,
246
+ * allocates nothing until first use, holds no audio itself — it only tracks who
247
+ * currently owns playback and how to stop them.
248
+ *
249
+ * Contract for holders (mirror of captureLock):
250
+ * - Stable id per path (constant for a singleton owner, `useId()` per instance).
251
+ * - `claimPlayback({ id, stop })` right before producing audio.
252
+ * - `releasePlayback(id)` when playback ends (finished, stopped, errored,
253
+ * unmounted). Release is id-guarded — a stale release is a safe no-op.
254
+ * - `stop` MUST be synchronous-effective (stop audio now).
255
+ *
256
+ * Subscribers get a `PlaybackTakeover` describing the holder change, which the
257
+ * AudioPlaybackHost uses to surface the Audio panel whenever playback gets
258
+ * "complex" (a cross-path takeover).
259
+ */
260
+ interface PlaybackHolder {
261
+ /** Stable identifier for the claiming playback path / instance. */
262
+ id: string;
263
+ /** Optional human label for diagnostics / the panel indicator. */
264
+ label?: string;
265
+ /** Stop audio immediately. Called when another path takes over. */
266
+ stop: () => void;
267
+ }
268
+ interface PlaybackTakeover {
269
+ /** The holder that now owns playback, or null when playback went idle. */
270
+ current: PlaybackHolder | null;
271
+ /** The holder that was stopped to grant ownership, if this was a takeover. */
272
+ preempted: PlaybackHolder | null;
273
+ }
274
+ /**
275
+ * Claim exclusive playback (start-always-wins). If a different holder currently
276
+ * owns playback, its `stop()` runs first (synchronously) so two outputs can
277
+ * never overlap. Re-claiming with the SAME id just refreshes the registration.
278
+ */
279
+ declare function claimPlayback(holder: PlaybackHolder): void;
280
+ /**
281
+ * Release playback for `id`. Id-guarded: if `id` is not the current owner (it
282
+ * was already taken over), this is a no-op.
283
+ */
284
+ declare function releasePlayback(id: string): void;
285
+ /** The id of the path currently holding playback, or null if idle. */
286
+ declare function getActivePlaybackHolderId(): string | null;
287
+ /** True while any path holds playback. */
288
+ declare function isPlaybackHeld(): boolean;
289
+ /** Subscribe to playback ownership changes. Returns an unsubscribe fn. */
290
+ declare function subscribePlaybackLock(cb: (event: PlaybackTakeover) => void): () => void;
291
+
292
+ /**
293
+ * unlock.ts — THE iOS/WebKit audio-output unlock primitive.
294
+ *
295
+ * WHY THIS EXISTS (mobile-silence class, 2026-08-30)
296
+ * --------------------------------------------------
297
+ * On iOS every browser is WebKit (Chrome and Firefox on iPhone included), and
298
+ * WebKit enforces two rules desktop browsers don't:
299
+ *
300
+ * 1. An `AudioContext` created (or resumed) OUTSIDE a user gesture starts
301
+ * `suspended` and stays silent. Our TTS starts audio from websocket
302
+ * callbacks seconds after the tap, so every utterance scheduled into a
303
+ * fresh per-utterance context played into a suspended context — total
304
+ * silence on iPhone while desktop worked fine.
305
+ * 2. Web Audio output is muted by the ringer/silent switch unless the page
306
+ * declares a playback audio session (`navigator.audioSession.type =
307
+ * "playback"`, iOS 16.4+).
308
+ *
309
+ * THE FIX: capture a real user gesture ONCE, and inside it (synchronously)
310
+ * - declare the playback audio session,
311
+ * - create ONE shared `AudioContext`, resume it, and play a one-frame
312
+ * silent buffer through it (the classic unlock), and
313
+ * - prime ONE shared `HTMLAudioElement` with a muted play() so element
314
+ * playback (the catalog engine) is also user-activated.
315
+ *
316
+ * Consumers then REUSE the shared unlocked context/element instead of minting
317
+ * fresh ones outside a gesture:
318
+ * - `SinkAwarePlayer` schedules through `getUnlockedAudioContext()` when it
319
+ * exists (per-utterance GainNode; the context is never closed).
320
+ * - `catalogAdapter` plays through `getPrimedMediaElement()` when it exists.
321
+ *
322
+ * Two entry points, both idempotent and framework-free:
323
+ * - `installAudioUnlockListeners()` — capture-phase pointerdown/keydown
324
+ * listeners that unlock on the user's FIRST interaction anywhere. Mounted
325
+ * once at app root (AudioPlaybackHost), so by the time any audio plays,
326
+ * the page is almost always already unlocked.
327
+ * - `primeAudioOutput()` — explicit call at the top of every audio-starting
328
+ * gesture handler (speak(), the Listen actions, transport buttons) as the
329
+ * belt-and-suspenders for taps the global listener could miss.
330
+ *
331
+ * Everything feature-detects: on desktop this is a no-op-cost shared context
332
+ * that behaves identically to before.
333
+ *
334
+ * THE CATEGORY IS SEQUENCED, NOT SHARED (2026-09-17)
335
+ * -------------------------------------------------
336
+ * `"playback"` is the OUTPUT category, and WebKit refuses every
337
+ * `getUserMedia({audio})` under it ("AudioSession category is not compatible
338
+ * with audio capture" — the /chat voice-input failure on iPhone). Capture needs
339
+ * `"play-and-record"`. This file never touches capture: the package's mic
340
+ * manager (`@ai-matrx/browser-audio` ≥ 0.4.0) switches the category right
341
+ * before its own capture and restores whatever this file declared once the
342
+ * microphone stops. Never make this file "smarter" about capture, and never
343
+ * declare the category from a recording surface — one owner per direction.
344
+ */
345
+ /**
346
+ * The ONE shared output context, unlocked by a user gesture. Null until the
347
+ * first `primeAudioOutput()` (or global-listener) gesture ran.
348
+ */
349
+ declare function getUnlockedAudioContext(): AudioContext | null;
350
+ /**
351
+ * The ONE shared media element, user-activated by a muted play(). Null until
352
+ * primed. Reused by element-based playback so `.play()` outside a gesture is
353
+ * allowed on WebKit (activation is per-element).
354
+ */
355
+ declare function getPrimedMediaElement(): HTMLAudioElement | null;
356
+ /**
357
+ * Unlock audio output. MUST be called synchronously inside a user-gesture
358
+ * call stack to have its full effect; safe (and cheap) to call any time.
359
+ */
360
+ declare function primeAudioOutput(): void;
361
+ /**
362
+ * Install capture-phase first-interaction listeners that unlock audio on ANY
363
+ * tap/keypress. Idempotent; the listeners stay installed (re-priming after
364
+ * the OS suspends the page costs nothing and heals `interrupted` contexts).
365
+ */
366
+ declare function installAudioUnlockListeners(): void;
367
+
368
+ type SinkListener = (deviceId: string) => void;
369
+ /** True when `HTMLMediaElement.setSinkId` exists (Chrome/Firefox, not Safari). */
370
+ declare function mediaElementSinkSupported(): boolean;
371
+ /** True when `AudioContext.setSinkId` exists (Chromium only). */
372
+ declare function audioContextSinkSupported(): boolean;
373
+ /**
374
+ * True when the browser can route output at all (either API). When false, the
375
+ * speaker picker is disabled and the user is told to choose output in OS
376
+ * settings.
377
+ */
378
+ declare function outputSelectionSupported(): boolean;
379
+ /** Read the current preferred output deviceId ("" = system default). */
380
+ declare function getPreferredOutputDeviceId(): string;
381
+ /**
382
+ * Set the preferred output device. Notifies subscribers (InlineMediaRef
383
+ * re-applies to its live media elements; SinkAwarePlayer re-routes any live
384
+ * playback context mid-utterance). Idempotent — a no-op when unchanged.
385
+ */
386
+ declare function setPreferredOutputDeviceId(deviceId: string): void;
387
+ /** Subscribe to preferred-output changes. Returns an unsubscribe fn. */
388
+ declare function subscribeOutputDevice(listener: SinkListener): () => void;
389
+ /**
390
+ * Apply the current preferred sink to one `HTMLMediaElement`. Feature-detected;
391
+ * on Safari (no `setSinkId`) or "" (system default) it's a clean no-op. A real
392
+ * failure (e.g. the device vanished) is reported loudly, never swallowed — the
393
+ * element keeps playing on the default device.
394
+ */
395
+ declare function applySinkToMediaElement(el: HTMLMediaElement | null | undefined): Promise<void>;
396
+
397
+ /**
398
+ * The slice of the Cartesia `Source` contract the player actually consumes.
399
+ * Structural — `@cartesia/cartesia-js`'s `Source` satisfies it — so the
400
+ * player has no hard SDK type dependency and is unit-testable with a stub.
401
+ */
402
+ interface PlayableSource {
403
+ readonly sampleRate: number;
404
+ /** Read up to `dst.length` samples; resolves with the count actually read.
405
+ * A short read (< dst.length) signals the end of the source. */
406
+ read(dst: Float32Array): Promise<number>;
407
+ /** Number of samples covering `durationSecs` at this source's sample rate. */
408
+ durationToSampleCount(durationSecs: number): number;
409
+ }
410
+ /** Minimal AudioContext surface the player uses — injectable for tests. */
411
+ interface SinkAwareAudioContext {
412
+ readonly currentTime: number;
413
+ readonly state: AudioContextState;
414
+ readonly destination: AudioNode;
415
+ createBufferSource(): AudioBufferSourceNode;
416
+ createBuffer(channels: number, length: number, sampleRate: number): AudioBuffer;
417
+ suspend(): Promise<void>;
418
+ resume(): Promise<void>;
419
+ close(): Promise<void>;
420
+ /** Chromium-only; absent on Firefox/Safari. */
421
+ setSinkId?: (deviceId: string) => Promise<void>;
422
+ }
423
+ interface SinkAwarePlayerOptions {
424
+ /** Seconds of audio to buffer per scheduling chunk (same as the SDK). */
425
+ bufferDuration: number;
426
+ /**
427
+ * Test seam: build the per-utterance context. Defaults to
428
+ * `new AudioContext({ sampleRate })`.
429
+ */
430
+ createContext?: (sampleRate: number) => SinkAwareAudioContext;
431
+ }
432
+ declare class SinkAwarePlayer {
433
+ #private;
434
+ constructor({ bufferDuration, createContext }: SinkAwarePlayerOptions);
435
+ /**
436
+ * Play audio from a source. Resolves when the audio has finished playing.
437
+ *
438
+ * Prefers the app's shared gesture-unlocked context (required for sound on
439
+ * iOS/WebKit); otherwise builds a fresh AudioContext per call (matching the
440
+ * SDK). Either way the output is routed to the user's preferred device and
441
+ * re-routed live on device change. Web Audio resamples per-buffer, so the
442
+ * shared context's hardware rate plays 44.1k sources correctly.
443
+ */
444
+ play(source: PlayableSource): Promise<void>;
445
+ /** Suspend playback. Throws before the first play (contract parity). */
446
+ pause(): Promise<void>;
447
+ /** Resume suspended playback. Throws before the first play. */
448
+ resume(): Promise<void>;
449
+ /** Pause when running, resume otherwise. Throws before the first play. */
450
+ toggle(): Promise<void>;
451
+ /**
452
+ * Stop playback. Throws before the first play (contract parity); a repeat
453
+ * stop is a silent no-op.
454
+ *
455
+ * Owned-context mode closes the context (the SDK contract). Shared mode
456
+ * NEVER closes the app-wide unlocked context — it stops this utterance's
457
+ * scheduled sources and tears down its GainNode, and resumes the context
458
+ * if this utterance had paused it (a suspended shared context would mute
459
+ * the NEXT utterance too).
460
+ */
461
+ stop(): Promise<void>;
462
+ }
463
+
464
+ /**
465
+ * THE single way to open a Cartesia TTS websocket from a browser (moved from matrx-frontend
466
+ * lib/cartesia/connection.ts; token, failure report and toast arrive through configureSpeech).
467
+ *
468
+ * Cartesia v4 exposes a snake_case, event based websocket. The rest of the
469
+ * app intentionally keeps one small, stable streaming contract, so SDK
470
+ * upgrades stay contained here instead of leaking through every TTS hook.
471
+ */
472
+
473
+ /**
474
+ * A provider session failure carrying the server's verdict: `message` is the sentence to show
475
+ * (the provider's raw text when the report failed), and `retryable` says whether reconnecting helps.
476
+ */
477
+ declare class ProviderSessionError extends Error {
478
+ readonly name = "ProviderSessionError";
479
+ readonly provider: ProviderSessionFailure["provider"];
480
+ readonly retryable: boolean;
481
+ readonly errorType: string | null;
482
+ readonly reported: boolean;
483
+ constructor(failure: ProviderSessionFailure, verdict: ProviderSessionFailureVerdict | null, fallbackMessage: string, options?: {
484
+ cause?: unknown;
485
+ });
486
+ }
487
+ /** True when the provider rejected the token (refresh once and retry). */
488
+ declare function isCartesiaAuthError(error: unknown): boolean;
489
+ interface CartesiaTtsSocketOptions {
490
+ container?: string;
491
+ encoding?: string;
492
+ sampleRate?: number;
493
+ }
494
+ type CartesiaTtsVoice = {
495
+ mode: "id";
496
+ id: string;
497
+ };
498
+ interface CartesiaTtsRequest {
499
+ modelId: string;
500
+ transcript: string;
501
+ voice: CartesiaTtsVoice;
502
+ language?: SupportedLanguage;
503
+ contextId?: string;
504
+ continue?: boolean;
505
+ addTimestamps?: boolean;
506
+ addPhonemeTimestamps?: boolean;
507
+ maxBufferDelayMs?: number;
508
+ generationConfig?: GenerationConfig;
509
+ }
510
+ type MessageListener = (message: string) => void;
511
+ /** The structural source consumed by SinkAwarePlayer. */
512
+ declare class CartesiaAudioSource {
513
+ #private;
514
+ readonly sampleRate: number;
515
+ constructor(sampleRate: number);
516
+ durationToSampleCount(durationSecs: number): number;
517
+ push(bytes: Uint8Array): void;
518
+ finish(): void;
519
+ read(destination: Float32Array): Promise<number>;
520
+ }
521
+ interface CartesiaTtsResponse {
522
+ source: CartesiaAudioSource;
523
+ on(event: "message", listener: MessageListener): void;
524
+ }
525
+ interface CartesiaConnectionCtx {
526
+ on(event: "close", listener: () => void): void;
527
+ }
528
+ interface CartesiaTtsSocket {
529
+ send(request: CartesiaTtsRequest): Promise<CartesiaTtsResponse>;
530
+ continue(request: CartesiaTtsRequest): Promise<CartesiaTtsResponse>;
531
+ disconnect(): void;
532
+ }
533
+ /** Open a connected socket, refreshing a rejected broker token once. */
534
+ declare function connectCartesiaTts(options?: CartesiaTtsSocketOptions): Promise<{
535
+ ws: CartesiaTtsSocket;
536
+ ctx: CartesiaConnectionCtx;
537
+ }>;
538
+
539
+ /**
540
+ * Cartesia playback adapter.
541
+ *
542
+ * Imperative twin of `useCartesiaSpeaker.speak()`: token → WebSocket → send →
543
+ * SinkAwarePlayer. Lives outside React so the singleton `playbackQueue` can
544
+ * drive it. Output device routing is owned by the player itself — it applies
545
+ * the preferred speaker at context creation and re-routes mid-utterance on
546
+ * device change (see features/audio/sinkAwarePlayer.ts).
547
+ *
548
+ * Note: Cartesia "speed" is a synthesis-time parameter, so live rate changes are
549
+ * not supported (no `setRate`). The queue's global rate is captured into the
550
+ * synthesis `speed` at enqueue time by the consumer instead.
551
+ */
552
+
553
+ /** Cartesia generation config from the person's speed / emotion. */
554
+ declare function buildGenerationConfig(opts?: {
555
+ speed?: number | null;
556
+ volume?: number | null;
557
+ emotion?: string | null;
558
+ }): {
559
+ speed: number;
560
+ volume: number;
561
+ emotion?: string;
562
+ };
563
+ declare const cartesiaAdapter: PlaybackAdapter;
564
+
565
+ export { type ActivePlayback, CartesiaAudioSource, type CartesiaConnectionCtx, type CartesiaTtsRequest, type CartesiaTtsResponse, type CartesiaTtsSocket, type CartesiaTtsSocketOptions, type CartesiaTtsVoice, type EnqueueResult, type PlayableSource, type PlaybackAdapter, type PlaybackAdapterCallbacks, type PlaybackHolder, type PlaybackItem, type PlaybackItemStatus, type PlaybackProvider, type PlaybackRequest, type PlaybackSnapshot, type PlaybackTakeover, ProviderSessionError, type SinkAwareAudioContext, SinkAwarePlayer, type SinkAwarePlayerOptions, type SpeakRequest, type SpeakResult, type SpeechPorts, type SpeechVoiceSettings, applySinkToMediaElement, audioContextSinkSupported, buildGenerationConfig, cartesiaAdapter, claimPlayback, clearPlayback, configureSpeech, connectCartesiaTts, enqueuePlayback, getActivePlaybackHolderId, getPlaybackSnapshot, getPreferredOutputDeviceId, getPrimedMediaElement, getUnlockedAudioContext, installAudioUnlockListeners, isCartesiaAuthError, isPlaybackHeld, mediaElementSinkSupported, outputSelectionSupported, pausePlayback, playPlaybackItem, primeAudioOutput, releasePlayback, removePlaybackItem, resetPlaybackForTests, resetSpeechPortsForTests, resolveSpeakEngine, resumePlayback, setPlaybackRate, setPreferredOutputDeviceId, skipPlayback, speak, subscribeOutputDevice, subscribePlayback, subscribePlaybackLock };