@tanstack/ai-grok 0.6.7 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/dist/esm/adapters/image.js +36 -17
  2. package/dist/esm/adapters/image.js.map +1 -1
  3. package/dist/esm/adapters/summarize.js +51 -22
  4. package/dist/esm/adapters/summarize.js.map +1 -1
  5. package/dist/esm/adapters/text.js +25 -10
  6. package/dist/esm/adapters/text.js.map +1 -1
  7. package/dist/esm/adapters/transcription.d.ts +84 -0
  8. package/dist/esm/adapters/transcription.js +109 -0
  9. package/dist/esm/adapters/transcription.js.map +1 -0
  10. package/dist/esm/adapters/tts.d.ts +70 -0
  11. package/dist/esm/adapters/tts.js +137 -0
  12. package/dist/esm/adapters/tts.js.map +1 -0
  13. package/dist/esm/audio/transcription-provider-options.d.ts +41 -0
  14. package/dist/esm/audio/tts-provider-options.d.ts +42 -0
  15. package/dist/esm/index.d.ts +8 -2
  16. package/dist/esm/index.js +17 -2
  17. package/dist/esm/index.js.map +1 -1
  18. package/dist/esm/model-meta.d.ts +6 -0
  19. package/dist/esm/model-meta.js +22 -1
  20. package/dist/esm/model-meta.js.map +1 -1
  21. package/dist/esm/realtime/adapter.d.ts +21 -0
  22. package/dist/esm/realtime/adapter.js +816 -0
  23. package/dist/esm/realtime/adapter.js.map +1 -0
  24. package/dist/esm/realtime/index.d.ts +4 -0
  25. package/dist/esm/realtime/realtime-contract.d.ts +30 -0
  26. package/dist/esm/realtime/token.d.ts +22 -0
  27. package/dist/esm/realtime/token.js +73 -0
  28. package/dist/esm/realtime/token.js.map +1 -0
  29. package/dist/esm/realtime/types.d.ts +95 -0
  30. package/dist/esm/utils/audio.d.ts +23 -0
  31. package/dist/esm/utils/audio.js +171 -0
  32. package/dist/esm/utils/audio.js.map +1 -0
  33. package/dist/esm/utils/index.d.ts +1 -0
  34. package/package.json +6 -3
  35. package/src/adapters/image.ts +41 -19
  36. package/src/adapters/summarize.ts +56 -25
  37. package/src/adapters/text.ts +26 -9
  38. package/src/adapters/transcription.ts +233 -0
  39. package/src/adapters/tts.ts +260 -0
  40. package/src/audio/transcription-provider-options.ts +54 -0
  41. package/src/audio/tts-provider-options.ts +44 -0
  42. package/src/index.ts +50 -1
  43. package/src/model-meta.ts +54 -0
  44. package/src/realtime/adapter.ts +1215 -0
  45. package/src/realtime/index.ts +18 -0
  46. package/src/realtime/realtime-contract.ts +46 -0
  47. package/src/realtime/token.ts +131 -0
  48. package/src/realtime/types.ts +105 -0
  49. package/src/utils/audio.ts +217 -0
  50. package/src/utils/index.ts +1 -0
@@ -0,0 +1,1215 @@
1
+ import { resolveDebugOption } from '@tanstack/ai/adapter-internals'
2
+ import type {
3
+ AnyClientTool,
4
+ AudioVisualization,
5
+ RealtimeEvent,
6
+ RealtimeEventHandler,
7
+ RealtimeMessage,
8
+ RealtimeMode,
9
+ RealtimeSessionConfig,
10
+ RealtimeStatus,
11
+ RealtimeToken,
12
+ } from '@tanstack/ai'
13
+ import type { InternalLogger } from '@tanstack/ai/adapter-internals'
14
+ import type { RealtimeAdapter, RealtimeConnection } from './realtime-contract'
15
+ import type { GrokRealtimeOptions } from './types'
16
+
17
+ const GROK_REALTIME_URL = 'https://api.x.ai/v1/realtime'
18
+
19
+ /**
20
+ * Runtime-checked field readers for untyped server events. Replace the
21
+ * drive-by `event.X as string` / `event.X as Record<string, unknown>` casts
22
+ * with readers that return `undefined` when the shape doesn't match, so a
23
+ * malformed frame can't throw a TypeError inside `handleServerEvent`.
24
+ */
25
+ function readString(
26
+ obj: Record<string, unknown>,
27
+ key: string,
28
+ ): string | undefined {
29
+ const value = obj[key]
30
+ return typeof value === 'string' ? value : undefined
31
+ }
32
+
33
+ function readObject(
34
+ obj: Record<string, unknown>,
35
+ key: string,
36
+ ): Record<string, unknown> | undefined {
37
+ const value = obj[key]
38
+ return value && typeof value === 'object' && !Array.isArray(value)
39
+ ? (value as Record<string, unknown>)
40
+ : undefined
41
+ }
42
+
43
+ function readObjectArray(
44
+ obj: Record<string, unknown>,
45
+ key: string,
46
+ ): Array<Record<string, unknown>> | undefined {
47
+ const value = obj[key]
48
+ if (!Array.isArray(value)) return undefined
49
+ return value.filter(
50
+ (item): item is Record<string, unknown> =>
51
+ item !== null && typeof item === 'object' && !Array.isArray(item),
52
+ )
53
+ }
54
+
55
+ type RealtimeServerError = Error & {
56
+ code?: string
57
+ type?: string
58
+ param?: string
59
+ }
60
+
61
+ /**
62
+ * Creates a Grok realtime adapter for client-side use.
63
+ *
64
+ * Uses WebRTC for browser connections (default). Mirrors the OpenAI realtime
65
+ * adapter because xAI's Voice Agent API is OpenAI-realtime-compatible — the
66
+ * only differences are the endpoint URL and default model.
67
+ *
68
+ * @example
69
+ * ```typescript
70
+ * import { RealtimeClient } from '@tanstack/ai-client'
71
+ * import { grokRealtime } from '@tanstack/ai-grok'
72
+ *
73
+ * const client = new RealtimeClient({
74
+ * getToken: () => fetch('/api/realtime-token').then(r => r.json()),
75
+ * adapter: grokRealtime(),
76
+ * })
77
+ * ```
78
+ */
79
+ export function grokRealtime(
80
+ options: GrokRealtimeOptions = {},
81
+ ): RealtimeAdapter {
82
+ const connectionMode = options.connectionMode ?? 'webrtc'
83
+ const logger = resolveDebugOption(options.debug)
84
+
85
+ return {
86
+ provider: 'grok',
87
+
88
+ async connect(
89
+ token: RealtimeToken,
90
+ _clientTools?: ReadonlyArray<AnyClientTool>,
91
+ ): Promise<RealtimeConnection> {
92
+ const model = token.config.model ?? 'grok-voice-fast-1.0'
93
+ logger.request(`activity=realtime provider=grok model=${model}`, {
94
+ provider: 'grok',
95
+ model,
96
+ })
97
+
98
+ if (connectionMode === 'webrtc') {
99
+ return createWebRTCConnection(token, logger)
100
+ }
101
+ const error = new Error('WebSocket connection mode not yet implemented')
102
+ logger.errors('grok.realtime fatal', {
103
+ error,
104
+ source: 'grok.realtime',
105
+ })
106
+ throw error
107
+ },
108
+ }
109
+ }
110
+
111
+ /**
112
+ * Creates a WebRTC connection to xAI's realtime API.
113
+ */
114
+ async function createWebRTCConnection(
115
+ token: RealtimeToken,
116
+ logger: InternalLogger,
117
+ ): Promise<RealtimeConnection> {
118
+ const model = token.config.model ?? 'grok-voice-fast-1.0'
119
+ const eventHandlers = new Map<RealtimeEvent, Set<RealtimeEventHandler<any>>>()
120
+
121
+ const pc = new RTCPeerConnection()
122
+
123
+ let audioContext: AudioContext | null = null
124
+ let inputAnalyser: AnalyserNode | null = null
125
+ let outputAnalyser: AnalyserNode | null = null
126
+ let inputSource: MediaStreamAudioSourceNode | null = null
127
+ let outputSource: MediaStreamAudioSourceNode | null = null
128
+ let localStream: MediaStream | null = null
129
+
130
+ let audioElement: HTMLAudioElement | null = null
131
+
132
+ let dataChannel: RTCDataChannel | null = null
133
+
134
+ let currentMode: RealtimeMode = 'idle'
135
+ let currentMessageId: string | null = null
136
+
137
+ // Flipped by `teardownConnection`. Guards `sendEvent` so post-disconnect
138
+ // calls (e.g. a React `useEffect` cleanup flushing queued events) are
139
+ // logged and skipped instead of silently piling up in `pendingEvents`.
140
+ let isTornDown = false
141
+
142
+ // Outbound events queued while the data channel isn't yet open. Declared
143
+ // here (rather than next to `sendEvent`) so `teardownConnection` — which
144
+ // lives higher up and can run from the SDP-path catch before `sendEvent`
145
+ // is defined — can drain it without hitting the TDZ.
146
+ const pendingEvents: Array<Record<string, unknown>> = []
147
+
148
+ // Tracks whether we've sent the first session.update. On the first update
149
+ // we attach a default input_audio_transcription so the server will emit
150
+ // user transcripts unless the caller opts out via
151
+ // `providerOptions.inputAudioTranscription = null | false`.
152
+ let hasSentInitialSessionUpdate = false
153
+
154
+ // Size hints for the fallback buffers returned when an analyser isn't yet
155
+ // populated. We return a *fresh* `Uint8Array` on each call so a caller
156
+ // that draws into it (e.g. a canvas visualiser zeroing the buffer) can't
157
+ // mutate a shared module-level instance for every other consumer.
158
+ const FALLBACK_FREQUENCY_BIN_COUNT = 1024
159
+ const FALLBACK_TIME_DOMAIN_SIZE = 2048
160
+ const FALLBACK_TIME_DOMAIN_FILL = 128
161
+
162
+ function emit<TEvent extends RealtimeEvent>(
163
+ event: TEvent,
164
+ payload: Parameters<RealtimeEventHandler<TEvent>>[0],
165
+ ) {
166
+ const handlers = eventHandlers.get(event)
167
+ if (handlers) {
168
+ for (const handler of handlers) {
169
+ handler(payload)
170
+ }
171
+ }
172
+ }
173
+
174
+ dataChannel = pc.createDataChannel('oai-events')
175
+
176
+ let dataChannelOpened = false
177
+ let rejectDataChannelReady: ((reason: unknown) => void) | null = null
178
+ let dataChannelReadyTimeout: ReturnType<typeof setTimeout> | null = null
179
+
180
+ const dataChannelReady = new Promise<void>((resolve, reject) => {
181
+ rejectDataChannelReady = (reason) => {
182
+ if (dataChannelReadyTimeout !== null) {
183
+ clearTimeout(dataChannelReadyTimeout)
184
+ dataChannelReadyTimeout = null
185
+ }
186
+ // One-shot: null out so later state transitions don't reject twice.
187
+ rejectDataChannelReady = null
188
+ reject(reason)
189
+ }
190
+
191
+ dataChannelReadyTimeout = setTimeout(() => {
192
+ if (!dataChannelOpened) {
193
+ rejectDataChannelReady?.(
194
+ new Error(
195
+ 'Data channel did not open within 15000ms — aborting connection',
196
+ ),
197
+ )
198
+ }
199
+ }, 15000)
200
+
201
+ dataChannel!.onopen = () => {
202
+ dataChannelOpened = true
203
+ if (dataChannelReadyTimeout !== null) {
204
+ clearTimeout(dataChannelReadyTimeout)
205
+ dataChannelReadyTimeout = null
206
+ }
207
+ // Once resolved, rejecting is a no-op — null out so teardown paths
208
+ // don't attempt a redundant reject on an already-settled promise.
209
+ rejectDataChannelReady = null
210
+ flushPendingEvents()
211
+ emit('status_change', { status: 'connected' as RealtimeStatus })
212
+ resolve()
213
+ }
214
+ })
215
+
216
+ dataChannel.onmessage = (event) => {
217
+ try {
218
+ const message = JSON.parse(event.data)
219
+ const messageRecord: Record<string, unknown> =
220
+ message !== null && typeof message === 'object' ? message : {}
221
+ logger.provider(
222
+ `provider=grok direction=in type=${readString(messageRecord, 'type') ?? '<unknown>'}`,
223
+ { frame: messageRecord },
224
+ )
225
+ handleServerEvent(messageRecord)
226
+ } catch (parseErr) {
227
+ logger.errors('grok.realtime fatal', {
228
+ error: parseErr,
229
+ source: 'grok.realtime',
230
+ })
231
+ emit('error', {
232
+ error:
233
+ parseErr instanceof Error ? parseErr : new Error(String(parseErr)),
234
+ })
235
+ }
236
+ }
237
+
238
+ dataChannel.onerror = (error) => {
239
+ // Closing the peer connection cascades into `onerror`/`onclose` on the
240
+ // data channel. Once teardown has started, re-surfacing those as
241
+ // `emit('error')` is noise that confuses consumers (they just called
242
+ // `disconnect()` — they don't want an error toast for it).
243
+ if (isTornDown) return
244
+ logger.errors('grok.realtime fatal', {
245
+ error,
246
+ source: 'grok.realtime',
247
+ })
248
+ // RTCErrorEvent exposes a typed `.error`; fall back to the event type
249
+ // name, then to a string representation, so the emitted error message
250
+ // doesn't end up as "[object Event]".
251
+ // `onerror` always fires with an Event (often an RTCErrorEvent), so we
252
+ // can read it via the untyped helpers without first proving object-ness.
253
+ const errorRecord = error as unknown as Record<string, unknown>
254
+ const rtcError = readObject(errorRecord, 'error')
255
+ const msg =
256
+ (rtcError && readString(rtcError, 'message')) ?? (error.type || 'unknown')
257
+ const dcErr = new Error(`Data channel error: ${msg}`)
258
+ if (!dataChannelOpened) {
259
+ rejectDataChannelReady?.(dcErr)
260
+ }
261
+ emit('error', { error: dcErr })
262
+ }
263
+
264
+ dataChannel.onclose = () => {
265
+ // Same rationale as `onerror` above: `pc.close()` during teardown
266
+ // cascades to the data channel's `onclose`. If we've already started
267
+ // teardown, there's nothing to do here.
268
+ if (isTornDown) return
269
+ if (!dataChannelOpened) {
270
+ rejectDataChannelReady?.(new Error('Data channel closed before opening'))
271
+ }
272
+ }
273
+
274
+ pc.ontrack = (event) => {
275
+ if (event.track.kind === 'audio' && event.streams[0]) {
276
+ setupOutputAudioAnalysis(event.streams[0])
277
+ }
278
+ }
279
+
280
+ // `status_change` has a single source of truth: `onconnectionstatechange`
281
+ // (the higher-level aggregate state). `oniceconnectionstatechange` is
282
+ // responsible only for rejecting `dataChannelReady` on ICE failures so we
283
+ // surface them without waiting for the 15s timeout.
284
+ pc.onconnectionstatechange = () => {
285
+ const state = pc.connectionState
286
+ logger.provider(`provider=grok pc.connectionState=${state}`, {
287
+ state,
288
+ })
289
+ if (state === 'failed' || state === 'disconnected' || state === 'closed') {
290
+ // Suppress the `status_change` emission when teardown is in progress:
291
+ // the user-facing `disconnect()` already emits `status_change: 'idle'`
292
+ // and then calls `teardownConnection()` → `pc.close()`, which fires
293
+ // `onconnectionstatechange` with state === 'closed'. Without this
294
+ // guard listeners would see two `idle` events per disconnect.
295
+ if (!isTornDown) {
296
+ emit('status_change', {
297
+ status:
298
+ state === 'failed'
299
+ ? ('error' as RealtimeStatus)
300
+ : ('idle' as RealtimeStatus),
301
+ })
302
+ }
303
+ if (!dataChannelOpened) {
304
+ // Reject on any terminal-ish pre-open state so callers don't hang
305
+ // for the full 15s timeout. The reject is one-shot — subsequent
306
+ // state changes become no-ops via the null-out in
307
+ // `rejectDataChannelReady`.
308
+ const message =
309
+ state === 'failed'
310
+ ? `PeerConnection failed before data channel opened`
311
+ : `PeerConnection entered state '${state}' before data channel opened`
312
+ rejectDataChannelReady?.(new Error(message))
313
+ }
314
+ // Auto-teardown on `failed`: without this the mic track, pc, and
315
+ // AudioContext stay allocated after a fatal connection failure, so the
316
+ // browser's mic indicator stays on and the user sees a broken
317
+ // "connected mic" state. `closed` already means pc was torn down
318
+ // (usually by teardownConnection itself) so nothing extra to do.
319
+ // `disconnected` is transient per the WebRTC spec and may recover, so
320
+ // we leave resources in place. `teardownConnection` is idempotent so
321
+ // a subsequent consumer `disconnect()` remains safe.
322
+ if (state === 'failed' && !isTornDown) {
323
+ void teardownConnection()
324
+ }
325
+ }
326
+ }
327
+
328
+ pc.oniceconnectionstatechange = () => {
329
+ const state = pc.iceConnectionState
330
+ logger.provider(`provider=grok pc.iceConnectionState=${state}`, {
331
+ state,
332
+ })
333
+ if (
334
+ !dataChannelOpened &&
335
+ (state === 'failed' || state === 'closed' || state === 'disconnected')
336
+ ) {
337
+ const message =
338
+ state === 'failed'
339
+ ? `ICE connection failed before data channel opened`
340
+ : `ICE connection entered state '${state}' before data channel opened`
341
+ rejectDataChannelReady?.(new Error(message))
342
+ }
343
+ }
344
+
345
+ /**
346
+ * Tear down every resource we may have allocated so the mic/pc/audio
347
+ * nodes/audio element don't leak on a failed connect. Safe to call from
348
+ * any point after `new RTCPeerConnection()`; each branch null-guards and
349
+ * swallows errors because cascading closes (e.g. `pc.close()` closing the
350
+ * data channel implicitly) are expected.
351
+ *
352
+ * Shared between the SDP-path catch, the post-SDP catch, and (implicitly
353
+ * via idempotency) the `disconnect()` entry point.
354
+ */
355
+ async function teardownConnection() {
356
+ // Flip the teardown flag BEFORE any awaits so handlers that fire during
357
+ // `await audioContext.close()` (or any other async step below) can guard
358
+ // on it — otherwise a late `pc.onconnectionstatechange` or `pc.ontrack`
359
+ // can allocate new resources or re-emit `status_change: idle` after the
360
+ // user-facing `disconnect()` already emitted one.
361
+ isTornDown = true
362
+
363
+ // Drop any queued events the caller sent before the data channel opened
364
+ // up front. Without this they'd accumulate across reconnect attempts
365
+ // (each connect allocates a fresh closure, but a caller holding the old
366
+ // `connection` reference could otherwise keep appending forever). Done
367
+ // at the top — before the awaits below — so `sendEvent` calls racing
368
+ // with teardown don't push into a list we're about to drain.
369
+ pendingEvents.length = 0
370
+
371
+ // Clear the data-channel-open timeout / reject the readiness promise
372
+ // if it's still pending. `rejectDataChannelReady` is one-shot and nulls
373
+ // itself on first call, so calling it from `disconnect()` after a
374
+ // successful open is a no-op.
375
+ rejectDataChannelReady?.(
376
+ new Error('Connection torn down before data channel opened'),
377
+ )
378
+
379
+ if (localStream) {
380
+ for (const track of localStream.getTracks()) {
381
+ track.stop()
382
+ }
383
+ localStream = null
384
+ }
385
+
386
+ // Output audio (populated by `pc.ontrack` → setupOutputAudioAnalysis,
387
+ // which may have fired during SDP negotiation before we threw).
388
+ if (audioElement) {
389
+ try {
390
+ audioElement.pause()
391
+ } catch {
392
+ // ignore — element may already be unloaded
393
+ }
394
+ audioElement.srcObject = null
395
+ audioElement = null
396
+ }
397
+ if (outputSource) {
398
+ try {
399
+ outputSource.disconnect()
400
+ } catch {
401
+ // ignore
402
+ }
403
+ outputSource = null
404
+ }
405
+ if (outputAnalyser) {
406
+ try {
407
+ outputAnalyser.disconnect()
408
+ } catch {
409
+ // ignore
410
+ }
411
+ outputAnalyser = null
412
+ }
413
+
414
+ // Input audio (populated by setupInputAudioAnalysis after SDP).
415
+ if (inputSource) {
416
+ try {
417
+ inputSource.disconnect()
418
+ } catch {
419
+ // ignore
420
+ }
421
+ inputSource = null
422
+ }
423
+ if (inputAnalyser) {
424
+ try {
425
+ inputAnalyser.disconnect()
426
+ } catch {
427
+ // ignore
428
+ }
429
+ inputAnalyser = null
430
+ }
431
+
432
+ if (dataChannel) {
433
+ try {
434
+ dataChannel.close()
435
+ } catch {
436
+ // ignore — channel may already be closed by pc.close()
437
+ }
438
+ dataChannel = null
439
+ }
440
+
441
+ try {
442
+ pc.close()
443
+ } catch {
444
+ // ignore — pc may already be closed
445
+ }
446
+
447
+ if (audioContext) {
448
+ try {
449
+ await audioContext.close()
450
+ } catch {
451
+ // ignore — context may already be closed
452
+ }
453
+ audioContext = null
454
+ }
455
+ }
456
+
457
+ // xAI requires an audio track in the SDP offer, same as OpenAI realtime.
458
+ //
459
+ // This try/catch also covers `getUserMedia` failure (e.g. the user denies
460
+ // microphone permission). `pc` + `dataChannel` are already allocated above
461
+ // and the 15s `dataChannelReady` timeout is already armed, so we MUST
462
+ // teardown on failure here — otherwise they leak until the tab closes.
463
+ // `teardownConnection` is idempotent and null-safe (runs fine even if the
464
+ // mic was never acquired).
465
+ try {
466
+ try {
467
+ localStream = await navigator.mediaDevices.getUserMedia({
468
+ audio: {
469
+ echoCancellation: true,
470
+ noiseSuppression: true,
471
+ sampleRate: 24000,
472
+ },
473
+ })
474
+ } catch (error) {
475
+ logger.errors('grok.realtime fatal', {
476
+ error,
477
+ source: 'grok.realtime.getUserMedia',
478
+ })
479
+ // Re-throw with the descriptive message callers rely on. Teardown runs
480
+ // in the outer catch below.
481
+ throw new Error(
482
+ `Microphone access required for realtime voice: ${error instanceof Error ? error.message : error}`,
483
+ )
484
+ }
485
+
486
+ for (const track of localStream.getAudioTracks()) {
487
+ pc.addTrack(track, localStream)
488
+ }
489
+
490
+ const offer = await pc.createOffer()
491
+ await pc.setLocalDescription(offer)
492
+
493
+ const sdpResponse = await fetch(`${GROK_REALTIME_URL}?model=${model}`, {
494
+ method: 'POST',
495
+ headers: {
496
+ Authorization: `Bearer ${token.token}`,
497
+ 'Content-Type': 'application/sdp',
498
+ },
499
+ body: offer.sdp,
500
+ })
501
+
502
+ if (!sdpResponse.ok) {
503
+ const errorText = await sdpResponse.text()
504
+ const error = new Error(
505
+ `Failed to establish WebRTC connection: ${sdpResponse.status} - ${errorText}`,
506
+ )
507
+ logger.errors('grok.realtime fatal', {
508
+ error,
509
+ source: 'grok.realtime.sdp',
510
+ status: sdpResponse.status,
511
+ })
512
+ throw error
513
+ }
514
+
515
+ const answerSdp = await sdpResponse.text()
516
+ await pc.setRemoteDescription({ type: 'answer', sdp: answerSdp })
517
+ } catch (err) {
518
+ await teardownConnection()
519
+ throw err
520
+ }
521
+
522
+ // Second cleanup scope: after SDP succeeds we still have to set up input
523
+ // audio analysis and wait for the data channel to open. Both can fail
524
+ // (AudioContext allocation, 15s timeout, ICE failure, pc.close from the
525
+ // other end, etc.) and those failures must NOT leave the mic/pc/audio
526
+ // nodes running.
527
+ try {
528
+ setupInputAudioAnalysis(localStream)
529
+ await dataChannelReady
530
+ } catch (err) {
531
+ await teardownConnection()
532
+ throw err
533
+ }
534
+
535
+ function handleServerEvent(event: Record<string, unknown>) {
536
+ const type = readString(event, 'type')
537
+
538
+ switch (type) {
539
+ case 'session.created':
540
+ case 'session.updated':
541
+ break
542
+
543
+ case 'input_audio_buffer.speech_started':
544
+ currentMode = 'listening'
545
+ emit('mode_change', { mode: 'listening' })
546
+ break
547
+
548
+ case 'input_audio_buffer.speech_stopped':
549
+ currentMode = 'thinking'
550
+ emit('mode_change', { mode: 'thinking' })
551
+ break
552
+
553
+ case 'input_audio_buffer.committed':
554
+ break
555
+
556
+ case 'conversation.item.input_audio_transcription.completed': {
557
+ const transcript = readString(event, 'transcript')
558
+ if (transcript === undefined) break
559
+ emit('transcript', { role: 'user', transcript, isFinal: true })
560
+ break
561
+ }
562
+
563
+ case 'response.created':
564
+ // Reset message id so a tool-only response (which never emits
565
+ // response.output_item.added for a message) can't reuse the previous
566
+ // turn's id when `response.done` later inspects this flag.
567
+ currentMessageId = null
568
+ currentMode = 'thinking'
569
+ emit('mode_change', { mode: 'thinking' })
570
+ break
571
+
572
+ case 'response.output_item.added': {
573
+ const item = readObject(event, 'item')
574
+ if (item && readString(item, 'type') === 'message') {
575
+ const id = readString(item, 'id')
576
+ if (id !== undefined) currentMessageId = id
577
+ }
578
+ break
579
+ }
580
+
581
+ // xAI realtime per docs uses `response.output_audio_transcript.*`;
582
+ // accept the legacy OpenAI-realtime `response.audio_transcript.*` as
583
+ // an alias so this adapter stays compatible across protocol versions.
584
+ case 'response.output_audio_transcript.delta':
585
+ case 'response.audio_transcript.delta': {
586
+ const delta = readString(event, 'delta')
587
+ if (delta === undefined) break
588
+ emit('transcript', {
589
+ role: 'assistant',
590
+ transcript: delta,
591
+ isFinal: false,
592
+ })
593
+ break
594
+ }
595
+
596
+ case 'response.output_audio_transcript.done':
597
+ case 'response.audio_transcript.done': {
598
+ const transcript = readString(event, 'transcript')
599
+ if (transcript === undefined) break
600
+ emit('transcript', { role: 'assistant', transcript, isFinal: true })
601
+ break
602
+ }
603
+
604
+ // xAI realtime per docs uses `response.text.*`; accept the legacy
605
+ // OpenAI-realtime `response.output_text.*` as an alias.
606
+ case 'response.text.delta':
607
+ case 'response.output_text.delta': {
608
+ const delta = readString(event, 'delta')
609
+ if (delta === undefined) break
610
+ emit('transcript', {
611
+ role: 'assistant',
612
+ transcript: delta,
613
+ isFinal: false,
614
+ })
615
+ break
616
+ }
617
+
618
+ case 'response.text.done':
619
+ case 'response.output_text.done': {
620
+ const text = readString(event, 'text')
621
+ if (text === undefined) break
622
+ emit('transcript', {
623
+ role: 'assistant',
624
+ transcript: text,
625
+ isFinal: true,
626
+ })
627
+ break
628
+ }
629
+
630
+ // xAI realtime per docs uses `response.output_audio.*`; accept the
631
+ // legacy OpenAI-realtime `response.audio.*` as an alias.
632
+ case 'response.output_audio.delta':
633
+ case 'response.audio.delta':
634
+ if (currentMode !== 'speaking') {
635
+ currentMode = 'speaking'
636
+ emit('mode_change', { mode: 'speaking' })
637
+ }
638
+ break
639
+
640
+ case 'response.output_audio.done':
641
+ case 'response.audio.done':
642
+ break
643
+
644
+ case 'response.function_call_arguments.done': {
645
+ // Only `call_id` is valid for `sendToolResult` correlation. Falling
646
+ // back to `item_id` would produce a tool-call id the server doesn't
647
+ // recognise when the result is posted back, silently dropping the
648
+ // tool execution. If `call_id` is missing we surface an error event
649
+ // so the UI can react instead of pretending the tool call succeeded.
650
+ const callId = readString(event, 'call_id')
651
+ const name = readString(event, 'name') ?? ''
652
+ const args = readString(event, 'arguments') ?? ''
653
+ if (!callId) {
654
+ logger.errors(
655
+ 'grok.realtime tool_call missing call_id — dropping tool_call',
656
+ {
657
+ source: 'grok.realtime',
658
+ event_type: 'response.function_call_arguments.done',
659
+ item_id: event.item_id,
660
+ },
661
+ )
662
+ emit('error', {
663
+ error: new Error(
664
+ 'Realtime tool call missing call_id; tool will not execute',
665
+ ),
666
+ })
667
+ break
668
+ }
669
+ try {
670
+ const input = JSON.parse(args)
671
+ emit('tool_call', { toolCallId: callId, toolName: name, input })
672
+ } catch {
673
+ emit('tool_call', { toolCallId: callId, toolName: name, input: args })
674
+ }
675
+ break
676
+ }
677
+
678
+ case 'response.done': {
679
+ const response = readObject(event, 'response') ?? {}
680
+ const output = readObjectArray(response, 'output')
681
+
682
+ // Only transition back to `listening` if the user hasn't already
683
+ // stopped capture — otherwise we'd override their explicit `idle`
684
+ // state and re-arm the mic visualisation.
685
+ if (currentMode !== 'idle') {
686
+ currentMode = 'listening'
687
+ emit('mode_change', { mode: 'listening' })
688
+ }
689
+
690
+ if (currentMessageId) {
691
+ const message: RealtimeMessage = {
692
+ id: currentMessageId,
693
+ role: 'assistant',
694
+ timestamp: Date.now(),
695
+ parts: [],
696
+ }
697
+
698
+ for (const item of output ?? []) {
699
+ if (readString(item, 'type') !== 'message') continue
700
+ const content = readObjectArray(item, 'content')
701
+ if (!content) continue
702
+ for (const part of content) {
703
+ const partType = readString(part, 'type')
704
+ if (partType === 'audio') {
705
+ const transcript = readString(part, 'transcript')
706
+ if (transcript) {
707
+ message.parts.push({ type: 'audio', transcript })
708
+ }
709
+ } else if (partType === 'text') {
710
+ const content = readString(part, 'text')
711
+ if (content) {
712
+ message.parts.push({ type: 'text', content })
713
+ }
714
+ }
715
+ }
716
+ }
717
+
718
+ emit('message_complete', { message })
719
+ currentMessageId = null
720
+ }
721
+ break
722
+ }
723
+
724
+ case 'conversation.item.truncated':
725
+ // Assistant playback was interrupted — flip mode back to `listening`
726
+ // unless the user already called `stopAudioCapture()` (idle). Without
727
+ // this the visualisation would stay stuck on `speaking` even though
728
+ // no audio is playing.
729
+ if (currentMode !== 'idle') {
730
+ currentMode = 'listening'
731
+ emit('mode_change', { mode: 'listening' })
732
+ }
733
+ emit('interrupted', { messageId: currentMessageId ?? undefined })
734
+ break
735
+
736
+ case 'error': {
737
+ // The realtime server's `error` envelope isn't guaranteed to carry
738
+ // an `error` object at all (network-layer corruption, protocol
739
+ // drift, etc.). Validate shape before dereferencing so a malformed
740
+ // payload can't throw a TypeError inside this handler and stop the
741
+ // switch from running for the rest of the session.
742
+ const errorObj = readObject(event, 'error') ?? {}
743
+ const message =
744
+ readString(errorObj, 'message') ?? 'Unknown realtime server error'
745
+ const err: RealtimeServerError = new Error(message)
746
+ // Preserve `code` / `type` / `param` on the Error as extra props so
747
+ // consumers can branch on them without re-parsing the raw event.
748
+ const code = readString(errorObj, 'code')
749
+ if (code !== undefined) err.code = code
750
+ const errType = readString(errorObj, 'type')
751
+ if (errType !== undefined) err.type = errType
752
+ const param = readString(errorObj, 'param')
753
+ if (param !== undefined) err.param = param
754
+ logger.errors('grok.realtime server error', {
755
+ ...errorObj,
756
+ source: 'grok.realtime server',
757
+ })
758
+ emit('error', { error: err })
759
+ break
760
+ }
761
+
762
+ default:
763
+ // The xAI realtime protocol is a moving target; log unhandled event
764
+ // types at provider level so they're visible during debugging without
765
+ // emitting a user-visible error.
766
+ logger.provider('grok.realtime unhandled server event', {
767
+ type: event.type,
768
+ })
769
+ break
770
+ }
771
+ }
772
+
773
+ function setupOutputAudioAnalysis(stream: MediaStream) {
774
+ // Bail out if teardown has already started. `pc.ontrack` can fire
775
+ // asynchronously after `teardownConnection()` has flipped `isTornDown`
776
+ // (e.g. a remote track arriving mid-close); without this guard we'd
777
+ // allocate a fresh AudioContext / audio element that nothing would ever
778
+ // clean up.
779
+ if (isTornDown) return
780
+
781
+ // Tear down any prior output audio before allocating new resources.
782
+ // `pc.ontrack` can fire multiple times over the lifetime of a session
783
+ // (e.g. after renegotiation), and without this we'd leak audio elements
784
+ // and analyser nodes.
785
+ if (audioElement) {
786
+ try {
787
+ audioElement.pause()
788
+ } catch {
789
+ // ignore — element may already be unloaded
790
+ }
791
+ audioElement.srcObject = null
792
+ audioElement = null
793
+ }
794
+ if (outputSource) {
795
+ try {
796
+ outputSource.disconnect()
797
+ } catch {
798
+ // ignore — may already be disconnected
799
+ }
800
+ outputSource = null
801
+ }
802
+ if (outputAnalyser) {
803
+ try {
804
+ outputAnalyser.disconnect()
805
+ } catch {
806
+ // ignore
807
+ }
808
+ outputAnalyser = null
809
+ }
810
+
811
+ audioElement = new Audio()
812
+ audioElement.srcObject = stream
813
+ audioElement.autoplay = true
814
+ audioElement.play().catch((e) => {
815
+ // Autoplay is commonly blocked until the user interacts with the page
816
+ // (browser gesture requirement). Surfacing this as a fatal `error`
817
+ // event makes the UI render a red/error state even though the
818
+ // connection is healthy — the page just needs a click. Log at a
819
+ // dedicated source tag so it's debuggable, but don't emit `error`.
820
+ logger.errors('grok.realtime audio autoplay blocked', {
821
+ error: e,
822
+ source: 'grok.realtime.audio_permission_required',
823
+ })
824
+ })
825
+
826
+ if (!audioContext) {
827
+ audioContext = new AudioContext()
828
+ }
829
+
830
+ if (audioContext.state === 'suspended') {
831
+ audioContext.resume().catch((err) => {
832
+ // Same rationale as the autoplay catch: `resume()` failure usually
833
+ // means the user hasn't interacted yet. Logging only — no error
834
+ // emit — so the UI doesn't go into a fatal state for a recoverable
835
+ // condition.
836
+ logger.errors('grok.realtime audioContext.resume failed', {
837
+ error: err,
838
+ source: 'grok.realtime',
839
+ })
840
+ })
841
+ }
842
+
843
+ outputAnalyser = audioContext.createAnalyser()
844
+ outputAnalyser.fftSize = 2048
845
+ outputAnalyser.smoothingTimeConstant = 0.3
846
+
847
+ outputSource = audioContext.createMediaStreamSource(stream)
848
+ outputSource.connect(outputAnalyser)
849
+ }
850
+
851
+ function setupInputAudioAnalysis(stream: MediaStream) {
852
+ // Defensive symmetry with `setupOutputAudioAnalysis`. Today this is
853
+ // only called inline after SDP negotiation, but keeping the guard
854
+ // means any future caller path (e.g. renegotiation) won't leak a fresh
855
+ // AudioContext after teardown.
856
+ if (isTornDown) return
857
+
858
+ if (!audioContext) {
859
+ audioContext = new AudioContext()
860
+ }
861
+
862
+ if (audioContext.state === 'suspended') {
863
+ audioContext.resume().catch((err) => {
864
+ // Same rationale as in setupOutputAudioAnalysis: a suspended
865
+ // AudioContext usually resumes after a user gesture. Log only —
866
+ // surfacing this as a fatal error makes the UI look broken for a
867
+ // recoverable condition.
868
+ logger.errors('grok.realtime audioContext.resume failed', {
869
+ error: err,
870
+ source: 'grok.realtime',
871
+ })
872
+ })
873
+ }
874
+
875
+ inputAnalyser = audioContext.createAnalyser()
876
+ inputAnalyser.fftSize = 2048
877
+ inputAnalyser.smoothingTimeConstant = 0.3
878
+
879
+ inputSource = audioContext.createMediaStreamSource(stream)
880
+ inputSource.connect(inputAnalyser)
881
+ }
882
+
883
+ function sendEvent(event: Record<string, unknown>) {
884
+ if (isTornDown) {
885
+ // The caller is holding onto a `connection` object after `disconnect()`
886
+ // (or a failed connect). Silently queueing would leak memory and the
887
+ // events would never flush. Log + drop so the misuse is visible in
888
+ // debug mode without escalating to a throw — throwing from a React
889
+ // useEffect cleanup path can break teardown ordering in the UI.
890
+ logger.errors('grok.realtime sendEvent after disconnect', {
891
+ eventType: readString(event, 'type') ?? '<unknown>',
892
+ source: 'grok.realtime',
893
+ })
894
+ return
895
+ }
896
+ if (dataChannel?.readyState === 'open') {
897
+ logger.provider(
898
+ `provider=grok direction=out type=${readString(event, 'type') ?? '<unknown>'}`,
899
+ { frame: event },
900
+ )
901
+ // Mirror the try/catch in `flushPendingEvents` — `dataChannel.send`
902
+ // can synchronously throw if the channel flipped to `closing` between
903
+ // our readyState check and this call, or if `JSON.stringify` chokes
904
+ // on a caller-supplied payload. Log + emit error instead of letting
905
+ // the exception propagate up through public `sendText` / `sendImage`
906
+ // / `updateSession` call sites.
907
+ try {
908
+ dataChannel.send(JSON.stringify(event))
909
+ } catch (error) {
910
+ logger.errors('grok.realtime sendEvent failed', {
911
+ error,
912
+ eventType: readString(event, 'type') ?? '<unknown>',
913
+ source: 'grok.realtime',
914
+ })
915
+ emit('error', {
916
+ error: error instanceof Error ? error : new Error(String(error)),
917
+ })
918
+ }
919
+ } else {
920
+ pendingEvents.push(event)
921
+ }
922
+ }
923
+
924
+ function flushPendingEvents() {
925
+ try {
926
+ for (const event of pendingEvents) {
927
+ logger.provider(
928
+ `provider=grok direction=out type=${readString(event, 'type') ?? '<unknown>'}`,
929
+ { frame: event },
930
+ )
931
+ dataChannel!.send(JSON.stringify(event))
932
+ }
933
+ pendingEvents.length = 0
934
+ } catch (error) {
935
+ // A send failure here (e.g. dataChannel went from 'open' back to
936
+ // 'closing' mid-flush, or JSON.stringify on a caller-provided event
937
+ // threw) would otherwise be silently swallowed. By the time we're
938
+ // called, `onopen` has already resolved `dataChannelReady`, so the
939
+ // consumer-facing signal is `emit('error')` — try rejectDataChannelReady
940
+ // as a defensive belt-and-braces in case this ever runs pre-resolve.
941
+ logger.errors('grok.realtime flushPendingEvents failed', {
942
+ error,
943
+ source: 'grok.realtime',
944
+ })
945
+ const err = error instanceof Error ? error : new Error(String(error))
946
+ rejectDataChannelReady?.(err)
947
+ emit('error', { error: err })
948
+ }
949
+ }
950
+
951
+ const connection: RealtimeConnection = {
952
+ async disconnect() {
953
+ // Reuse the same teardown path as the failed-connect branches so
954
+ // every cleanup site stays in sync (input analyser, output analyser,
955
+ // output source, audio element, etc.).
956
+ await teardownConnection()
957
+ emit('status_change', { status: 'idle' as RealtimeStatus })
958
+ },
959
+
960
+ async startAudioCapture() {
961
+ if (localStream) {
962
+ for (const track of localStream.getAudioTracks()) {
963
+ track.enabled = true
964
+ }
965
+ }
966
+ currentMode = 'listening'
967
+ emit('mode_change', { mode: 'listening' })
968
+ },
969
+
970
+ stopAudioCapture() {
971
+ if (localStream) {
972
+ for (const track of localStream.getAudioTracks()) {
973
+ track.enabled = false
974
+ }
975
+ }
976
+ currentMode = 'idle'
977
+ emit('mode_change', { mode: 'idle' })
978
+ },
979
+
980
+ sendText(text: string) {
981
+ sendEvent({
982
+ type: 'conversation.item.create',
983
+ item: {
984
+ type: 'message',
985
+ role: 'user',
986
+ content: [{ type: 'input_text', text }],
987
+ },
988
+ })
989
+ sendEvent({ type: 'response.create' })
990
+ },
991
+
992
+ sendImage(imageData: string, mimeType: string) {
993
+ // Accept:
994
+ // - http(s):// URLs → forward as-is
995
+ // - data: URIs (e.g. from FileReader.readAsDataURL) → forward as-is
996
+ // so we don't double-wrap into `data:image/png;base64,data:image/png;base64,…`
997
+ // - bare base64 → wrap in `data:${mimeType};base64,…`
998
+ const isAlreadyUrlOrDataUri =
999
+ imageData.startsWith('http://') ||
1000
+ imageData.startsWith('https://') ||
1001
+ imageData.startsWith('data:')
1002
+ const imageContent = {
1003
+ type: 'input_image',
1004
+ // The OpenAI-realtime content part (which this adapter mirrors) nests
1005
+ // the URL under an `image_url: { url: ... }` object, not a bare
1006
+ // string.
1007
+ image_url: {
1008
+ url: isAlreadyUrlOrDataUri
1009
+ ? imageData
1010
+ : `data:${mimeType};base64,${imageData}`,
1011
+ },
1012
+ }
1013
+
1014
+ sendEvent({
1015
+ type: 'conversation.item.create',
1016
+ item: {
1017
+ type: 'message',
1018
+ role: 'user',
1019
+ content: [imageContent],
1020
+ },
1021
+ })
1022
+ sendEvent({ type: 'response.create' })
1023
+ },
1024
+
1025
+ sendToolResult(callId: string, result: string) {
1026
+ sendEvent({
1027
+ type: 'conversation.item.create',
1028
+ item: {
1029
+ type: 'function_call_output',
1030
+ call_id: callId,
1031
+ output: result,
1032
+ },
1033
+ })
1034
+ sendEvent({ type: 'response.create' })
1035
+ },
1036
+
1037
+ updateSession(config: Partial<RealtimeSessionConfig>) {
1038
+ const sessionUpdate: Record<string, unknown> = {}
1039
+
1040
+ if (config.instructions) {
1041
+ sessionUpdate.instructions = config.instructions
1042
+ }
1043
+
1044
+ if (config.voice) {
1045
+ sessionUpdate.voice = config.voice
1046
+ }
1047
+
1048
+ if (config.vadMode) {
1049
+ if (config.vadMode === 'semantic') {
1050
+ sessionUpdate.turn_detection = {
1051
+ type: 'semantic_vad',
1052
+ eagerness: config.semanticEagerness ?? 'medium',
1053
+ }
1054
+ } else if (config.vadMode === 'server') {
1055
+ sessionUpdate.turn_detection = {
1056
+ type: 'server_vad',
1057
+ threshold: config.vadConfig?.threshold ?? 0.5,
1058
+ prefix_padding_ms: config.vadConfig?.prefixPaddingMs ?? 300,
1059
+ silence_duration_ms: config.vadConfig?.silenceDurationMs ?? 500,
1060
+ }
1061
+ } else {
1062
+ sessionUpdate.turn_detection = null
1063
+ }
1064
+ }
1065
+
1066
+ if (config.tools !== undefined) {
1067
+ sessionUpdate.tools = config.tools.map((t) => ({
1068
+ type: 'function',
1069
+ name: t.name,
1070
+ description: t.description,
1071
+ parameters: t.inputSchema ?? { type: 'object', properties: {} },
1072
+ }))
1073
+ sessionUpdate.tool_choice = 'auto'
1074
+ }
1075
+
1076
+ if (config.outputModalities) {
1077
+ sessionUpdate.modalities = config.outputModalities
1078
+ }
1079
+
1080
+ if (config.temperature !== undefined) {
1081
+ sessionUpdate.temperature = config.temperature
1082
+ }
1083
+
1084
+ if (config.maxOutputTokens !== undefined) {
1085
+ sessionUpdate.max_response_output_tokens = config.maxOutputTokens
1086
+ }
1087
+
1088
+ // Let callers forward an explicit `input_audio_transcription` value
1089
+ // through `providerOptions` — including `null` / `false` to disable
1090
+ // the feature. Only apply our `grok-stt` default on the first
1091
+ // session.update and only if the caller hasn't set it themselves.
1092
+ const providerOptions: Record<string, unknown> =
1093
+ config.providerOptions ?? {}
1094
+ const callerTranscription =
1095
+ 'inputAudioTranscription' in providerOptions
1096
+ ? providerOptions.inputAudioTranscription
1097
+ : 'input_audio_transcription' in providerOptions
1098
+ ? providerOptions.input_audio_transcription
1099
+ : undefined
1100
+ if (callerTranscription !== undefined) {
1101
+ sessionUpdate.input_audio_transcription =
1102
+ callerTranscription === false ? null : callerTranscription
1103
+ } else if (!hasSentInitialSessionUpdate) {
1104
+ sessionUpdate.input_audio_transcription = { model: 'grok-stt' }
1105
+ }
1106
+
1107
+ if (Object.keys(sessionUpdate).length > 0) {
1108
+ sendEvent({
1109
+ type: 'session.update',
1110
+ session: sessionUpdate,
1111
+ })
1112
+ hasSentInitialSessionUpdate = true
1113
+ }
1114
+ },
1115
+
1116
+ interrupt() {
1117
+ sendEvent({ type: 'response.cancel' })
1118
+ currentMode = 'listening'
1119
+ emit('mode_change', { mode: 'listening' })
1120
+ emit('interrupted', { messageId: currentMessageId ?? undefined })
1121
+ },
1122
+
1123
+ on<TEvent extends RealtimeEvent>(
1124
+ event: TEvent,
1125
+ handler: RealtimeEventHandler<TEvent>,
1126
+ ): () => void {
1127
+ if (!eventHandlers.has(event)) {
1128
+ eventHandlers.set(event, new Set())
1129
+ }
1130
+ eventHandlers.get(event)!.add(handler)
1131
+
1132
+ return () => {
1133
+ eventHandlers.get(event)?.delete(handler)
1134
+ }
1135
+ },
1136
+
1137
+ getAudioVisualization(): AudioVisualization {
1138
+ function calculateLevel(analyser: AnalyserNode): number {
1139
+ const data = new Uint8Array(analyser.fftSize)
1140
+ analyser.getByteTimeDomainData(data)
1141
+
1142
+ let maxDeviation = 0
1143
+ for (const sample of data) {
1144
+ const deviation = Math.abs(sample - 128)
1145
+ if (deviation > maxDeviation) {
1146
+ maxDeviation = deviation
1147
+ }
1148
+ }
1149
+
1150
+ const normalized = maxDeviation / 128
1151
+ return Math.min(1, normalized * 1.5)
1152
+ }
1153
+
1154
+ return {
1155
+ get inputLevel() {
1156
+ if (!inputAnalyser) return 0
1157
+ return calculateLevel(inputAnalyser)
1158
+ },
1159
+
1160
+ get outputLevel() {
1161
+ if (!outputAnalyser) return 0
1162
+ return calculateLevel(outputAnalyser)
1163
+ },
1164
+
1165
+ getInputFrequencyData() {
1166
+ if (!inputAnalyser)
1167
+ return new Uint8Array(FALLBACK_FREQUENCY_BIN_COUNT)
1168
+ const data = new Uint8Array(inputAnalyser.frequencyBinCount)
1169
+ inputAnalyser.getByteFrequencyData(data)
1170
+ return data
1171
+ },
1172
+
1173
+ getOutputFrequencyData() {
1174
+ if (!outputAnalyser)
1175
+ return new Uint8Array(FALLBACK_FREQUENCY_BIN_COUNT)
1176
+ const data = new Uint8Array(outputAnalyser.frequencyBinCount)
1177
+ outputAnalyser.getByteFrequencyData(data)
1178
+ return data
1179
+ },
1180
+
1181
+ getInputTimeDomainData() {
1182
+ if (!inputAnalyser)
1183
+ return new Uint8Array(FALLBACK_TIME_DOMAIN_SIZE).fill(
1184
+ FALLBACK_TIME_DOMAIN_FILL,
1185
+ )
1186
+ const data = new Uint8Array(inputAnalyser.fftSize)
1187
+ inputAnalyser.getByteTimeDomainData(data)
1188
+ return data
1189
+ },
1190
+
1191
+ getOutputTimeDomainData() {
1192
+ if (!outputAnalyser)
1193
+ return new Uint8Array(FALLBACK_TIME_DOMAIN_SIZE).fill(
1194
+ FALLBACK_TIME_DOMAIN_FILL,
1195
+ )
1196
+ const data = new Uint8Array(outputAnalyser.fftSize)
1197
+ outputAnalyser.getByteTimeDomainData(data)
1198
+ return data
1199
+ },
1200
+
1201
+ get inputSampleRate() {
1202
+ return 24000
1203
+ },
1204
+
1205
+ get outputSampleRate() {
1206
+ return 24000
1207
+ },
1208
+ }
1209
+ },
1210
+ }
1211
+
1212
+ // `dataChannelReady` was already awaited inside the post-SDP try/catch
1213
+ // above so we can short-circuit on failures with full teardown.
1214
+ return connection
1215
+ }