ai 7.0.101 → 7.0.102

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +30 -0
  2. package/dist/index.d.ts +82 -20
  3. package/dist/index.js +2010 -330
  4. package/dist/index.js.map +1 -1
  5. package/dist/internal/index.js +1 -1
  6. package/docs/03-agents/07-workflow-agent.mdx +1 -1
  7. package/docs/03-ai-sdk-core/36-realtime.mdx +41 -18
  8. package/docs/04-ai-sdk-ui/21-transport.mdx +1 -1
  9. package/docs/07-reference/02-ai-sdk-ui/05-use-realtime.mdx +364 -48
  10. package/docs/07-reference/04-ai-sdk-workflow/02-workflow-chat-transport.mdx +4 -4
  11. package/docs/07-reference/05-ai-sdk-errors/index.mdx +131 -36
  12. package/package.json +12 -12
  13. package/src/generate-text/execute-tools-from-stream.ts +7 -0
  14. package/src/generate-text/stream-text.ts +3 -1
  15. package/src/realtime/__fixtures__/fake-live-websocket.ts +71 -0
  16. package/src/realtime/__fixtures__/fake-realtime.ts +36 -0
  17. package/src/realtime/__fixtures__/fake-webrtc.ts +133 -0
  18. package/src/realtime/browser-realtime-audio.ts +107 -10
  19. package/src/realtime/browser-realtime-live-websocket.ts +247 -0
  20. package/src/realtime/browser-realtime-transport.ts +235 -69
  21. package/src/realtime/browser-realtime-webrtc.ts +582 -0
  22. package/src/realtime/encode-realtime-frame.ts +33 -0
  23. package/src/realtime/index.ts +1 -0
  24. package/src/realtime/realtime-attempt.ts +45 -0
  25. package/src/realtime/realtime-command-tracker.ts +81 -0
  26. package/src/realtime/realtime-event-channel.ts +170 -0
  27. package/src/realtime/realtime-event-reducer.ts +3 -0
  28. package/src/realtime/realtime-session-state.ts +65 -0
  29. package/src/realtime/realtime-session.ts +768 -218
  30. package/src/realtime/realtime-types.ts +1 -1
  31. package/src/realtime/validate-realtime-setup.ts +37 -0
@@ -12,26 +12,218 @@ description: API reference for the experimental_useRealtime hook.
12
12
  Creates a browser-side realtime session for bidirectional audio and text
13
13
  conversations with a realtime provider model.
14
14
 
15
- The hook connects to a realtime WebSocket using a short-lived token from your
16
- setup endpoint, returns messages as `UIMessage[]`, and provides controls for
17
- audio capture, playback, text input, and tool output.
15
+ The hook supports token-based provider WebSockets, application-owned WebSocket
16
+ relays, and optional WebRTC where the model declares support. It provides controls for capture and playback, plus turn-based text input
17
+ and tool output. Turn-based conversation messages use `UIMessage[]`; continuous
18
+ transcript fragments remain in `session`.
18
19
 
19
20
  ```tsx
20
21
  import { openai } from '@ai-sdk/openai';
21
22
  import { experimental_useRealtime } from '@ai-sdk/react';
22
23
 
23
- const realtime = experimental_useRealtime({
24
- model: openai.experimental_realtime('gpt-realtime'),
25
- api: {
26
- token: '/api/realtime/setup',
27
- },
28
- });
24
+ const model = openai.experimental_realtime('gpt-realtime');
25
+
26
+ function Conversation() {
27
+ const realtime = experimental_useRealtime({
28
+ model,
29
+ api: { token: '/api/realtime/setup' },
30
+ });
31
+ return <button onClick={realtime.connect}>Connect</button>;
32
+ }
29
33
  ```
30
34
 
31
35
  For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
32
36
  `api.token` at a server-side setup endpoint that calls
33
37
  `gateway.experimental_realtime.getToken()`.
34
38
 
39
+ Keep the model and session configuration stable across renders, using module scope
40
+ or `useMemo`. Replacing either object replaces the hook's session.
41
+
42
+ ## Continuous conversations
43
+
44
+ For OpenAI Live, use `openai.experimental_realtime('gpt-live-1')` and an application-owned WebSocket
45
+ relay. The relay supplies server credentials and forwards the provider's native
46
+ text frames; it must authenticate clients before accepting connections.
47
+ Use `wss:` in production, keep provider credentials server-side, and apply your
48
+ application's authentication to the relay connection.
49
+
50
+ ```tsx
51
+ import {
52
+ openai,
53
+ type Experimental_OpenAIRealtimeModelLiveOptions,
54
+ } from '@ai-sdk/openai';
55
+ import { experimental_useRealtime } from '@ai-sdk/react';
56
+
57
+ const model = openai.experimental_realtime('gpt-live-1');
58
+ const sessionConfig = {
59
+ instructions: 'Be a concise, friendly assistant.',
60
+ providerOptions: {
61
+ openai: {
62
+ delegation: { type: 'client' },
63
+ } satisfies Experimental_OpenAIRealtimeModelLiveOptions,
64
+ },
65
+ };
66
+
67
+ function LiveConversation() {
68
+ const realtime = experimental_useRealtime({
69
+ model,
70
+ api: { websocket: 'wss://your-app.example/live' },
71
+ sessionConfig,
72
+ });
73
+ return (
74
+ <>
75
+ <button onClick={realtime.connect}>Connect microphone</button>
76
+ <button onClick={() => realtime.close()}>End conversation</button>
77
+ <p>
78
+ {realtime.status}: {realtime.session?.usage?.seconds} seconds
79
+ </p>
80
+ </>
81
+ );
82
+ }
83
+ ```
84
+
85
+ The current continuous browser runtime supports a JSON/PCM16 WebSocket relay media
86
+ profile. Continuous conversation semantics alone do not guarantee support for every
87
+ codec or transport. Applications with their own audio pipeline can use the low-level
88
+ provider for other supported codecs. OpenAI-specific settings
89
+ live under `providerOptions.openai` and use camelCase fields; the provider converts
90
+ them to wire names.
91
+
92
+ ### Application-handled client delegation
93
+
94
+ Live reports client delegation metadata through `session.delegations` and normalized
95
+ events. The application owns any text conversation, agent context, tool execution,
96
+ and result validation. The SDK does not run a generic agent executor for Live, and
97
+ Live delegations do not invoke `onToolCall`.
98
+
99
+ Use `sendEvent` to append context or a result. Set `delegationId` to a known client
100
+ delegation from the current session, or `null` for session-wide context:
101
+
102
+ ```tsx
103
+ await realtime.sendEvent({
104
+ type: 'context-append',
105
+ delegationId: null,
106
+ content: 'The application has confirmed that the appointment is at 3 PM.',
107
+ providerOptions: { openai: { channel: 'commentary' } },
108
+ });
109
+ ```
110
+
111
+ The example uses application-provided content, not task arguments inferred from
112
+ delegation metadata. OpenAI's context channels are `commentary`, `thinking`, and
113
+ `instructions`; reserve `instructions` for trusted application instructions. A
114
+ successful send confirms local submission, not provider acceptance or audible delivery.
115
+
116
+ ### Optional WebRTC
117
+
118
+ For browser-direct Live audio, replace `api.websocket` with
119
+ `api: { session: '/api/realtime-live' }`. The hook posts JSON `{ sdp, sessionConfig }`
120
+ to that application endpoint and expects JSON `{ sdp, sessionId }` back. On the
121
+ server, authenticate the application user and validate the offer and allowed settings.
122
+
123
+ Use a same-origin broker authenticated with your application's session cookie.
124
+ The built-in setup request uses the browser's default same-origin credentials;
125
+ `api.session` does not accept custom authorization headers, a custom fetch, or
126
+ cross-origin credential options. A bearer-token-only or cross-origin cookie broker
127
+ therefore needs an application-owned same-origin endpoint in front of it. The
128
+ answer must contain nonempty `sdp` and `sessionId` strings (after trimming for
129
+ validation), and the JSON response body is limited to 1 MiB.
130
+
131
+ Exchange SDP using the server-held provider key:
132
+
133
+ ```ts
134
+ const answer = await openai
135
+ .experimental_realtime('gpt-live-1')
136
+ .doCreateWebRTCSession({
137
+ sdp,
138
+ sessionConfig: {
139
+ providerOptions: {
140
+ openai: {
141
+ delegation: { type: 'client' },
142
+ client: {
143
+ dataChannel: {
144
+ allowedClientEvents: ['session.close', 'session.thinking.append'],
145
+ allowedServerEvents: [
146
+ { type: 'session.started' },
147
+ { type: 'session.closed' },
148
+ { type: 'session.usage.updated' },
149
+ { type: 'session.delegation.created' },
150
+ { type: 'session.input_transcript.delta' },
151
+ { type: 'session.output_transcript.delta' },
152
+ { type: 'session.thinking.appended' },
153
+ { type: 'error' },
154
+ ],
155
+ },
156
+ },
157
+ },
158
+ },
159
+ },
160
+ });
161
+ // Return Response.json(answer) from the application endpoint.
162
+ ```
163
+
164
+ Permissions are server-owned; override browser-supplied delegation and data-channel
165
+ policy. Allow the lifecycle, caption, delegation, command acknowledgment, and error
166
+ events needed by your UI. The runnable `/realtime-live` example in
167
+ `examples/ai-e2e-next` adds protocol mute and all three context channels.
168
+
169
+ WebRTC negotiates audio through SDP, so omit fixed audio formats and the PCM-only
170
+ `maxPlaybackBufferSeconds`. Audio travels as media rather than JSON audio commands.
171
+ `connect({ capture: false })` starts without microphone capture and keeps a reusable
172
+ audio sender; `resumeAudioCapture()` can attach a microphone later. Microphone access
173
+ requires browser permission and a secure context. Autoplay may require a user gesture
174
+ followed by `resumePlayback()`.
175
+
176
+ One live audio track is selected per attachment, preferring an enabled, unmuted
177
+ track. `isCapturing` follows that sender track, including mute and ended events;
178
+ it never switches tracks automatically. Caller-owned tracks are detached without
179
+ being stopped or disabled. Capture controls serialize sender changes; a stop is
180
+ reported only after detachment succeeds or the peer closes. If detachment fails,
181
+ the peer is closed to stop transmission while leaving borrowed tracks untouched.
182
+ Client delegation remains application-handled on either transport, and Live session
183
+ updates remain unsupported.
184
+
185
+ ## Lifecycle and failure handling
186
+
187
+ Use `status === 'connected'` as the readiness signal; `connect()` does not promise
188
+ to wait for provider readiness. Operational startup errors are reported through
189
+ `onError` and `status`; the legacy resolve-and-report behavior is preserved.
190
+
191
+ `close()` stops local capture and submissions, then drains events until the provider
192
+ confirms final usage or the close deadline expires. A failed close-command send
193
+ uses the shorter accepted-event drain rather than waiting for an acknowledgment
194
+ to an unsent command. Read `session.finalization`: a fulfilled close promise does
195
+ not by itself confirm usage. `disconnect()` and unmount release resources immediately.
196
+
197
+ Hook controls keep stable identities across renders and provider events. A retained
198
+ control targets the current **committed** session, including after a model or endpoint
199
+ change. Uncommitted renders cannot replace the active session or its callbacks.
200
+ Callback-only updates take effect at commit without reconnecting.
201
+
202
+ After unmount, `connect`, `close`, `resumeAudioCapture`, and `resumePlayback` reject
203
+ with a mounted-hook error. Other controls throw that error synchronously, including
204
+ `sendEvent`, which preserves synchronous validation and returns a promise for an
205
+ accepted submission. Retained controls cannot reopen an unmounted session.
206
+
207
+ Command rejection and playback failures are recoverable and do not mark a healthy
208
+ protocol connection as failed. Irrecoverable transport failures and protocol queue
209
+ overflow stop submissions and capture, drain the accepted event prefix, and clean
210
+ up. Events are never arbitrarily dropped while continuing with unreliable state.
211
+ No commands or side-effecting tools are transparently replayed on a replacement
212
+ connection.
213
+
214
+ Continuous Live WebSocket sessions have a 128 KiB outgoing wire-frame limit and a
215
+ 128 KiB combined buffered-send budget, including the next frame and JSON/base64
216
+ encoding overhead. Oversized control messages are rejected, not silently split into
217
+ multiple commands. Keep Live context submissions within this budget; the example
218
+ relay's 128 KiB `maxPayload` aligns with the frame limit. These byte caps do not apply
219
+ to legacy turn-based sessions, which retain their pre-Live unlimited byte policy,
220
+ or to the optional WebRTC transport.
221
+
222
+ The continuous WebSocket PCM playback budget bounds local latency and memory. On overflow, stale queued audio
223
+ is discarded, playback pauses, and `onError` reports an audible gap. The connection
224
+ stays alive; `resumePlayback()` resumes from fresh audio at the live edge. Captions
225
+ and completion of delegated work do not prove that the corresponding audio was heard.
226
+
35
227
  ## Import
36
228
 
37
229
  <Snippet
@@ -52,9 +244,9 @@ For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
52
244
  },
53
245
  {
54
246
  name: 'api',
55
- type: '{ token: string }',
247
+ type: '{ token: string } | { websocket: string; protocols?: string[] } | { session: string }',
56
248
  description:
57
- 'API endpoints used by the realtime session. The token endpoint is called with POST to create the realtime setup response.',
249
+ 'Choose one supported connection mechanism: token setup, a WebSocket relay, or a WebRTC SDP endpoint. Conversation semantics and connection capabilities are independent.',
58
250
  properties: [
59
251
  {
60
252
  type: 'Object',
@@ -62,9 +254,31 @@ For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
62
254
  {
63
255
  name: 'token',
64
256
  type: 'string',
257
+ isOptional: true,
65
258
  description:
66
259
  'The setup endpoint that returns an Experimental_RealtimeSetupResponse.',
67
260
  },
261
+ {
262
+ name: 'session',
263
+ type: 'string',
264
+ isOptional: true,
265
+ description:
266
+ 'Application WebRTC setup endpoint. Accepts JSON { sdp, sessionConfig } and returns JSON { sdp, sessionId }.',
267
+ },
268
+ {
269
+ name: 'websocket',
270
+ type: 'string',
271
+ isOptional: true,
272
+ description:
273
+ 'Application-owned raw-protocol relay URL. Use wss in production; never put provider credentials in the URL.',
274
+ },
275
+ {
276
+ name: 'protocols',
277
+ type: 'string[]',
278
+ isOptional: true,
279
+ description:
280
+ 'Optional relay subprotocols, supported only with websocket.',
281
+ },
68
282
  ],
69
283
  },
70
284
  ],
@@ -90,12 +304,40 @@ For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
90
304
  description:
91
305
  'Maximum number of provider events to keep in the events array. Defaults to 500.',
92
306
  },
307
+ {
308
+ name: 'startupTimeoutMs',
309
+ type: 'number',
310
+ isOptional: true,
311
+ description:
312
+ 'Readiness deadline, including transport setup. Defaults to 30000.',
313
+ },
314
+ {
315
+ name: 'closeTimeoutMs',
316
+ type: 'number',
317
+ isOptional: true,
318
+ description:
319
+ 'Deadline for confirmed provider finalization after graceful close. Defaults to 15000.',
320
+ },
321
+ {
322
+ name: 'rtcDisconnectTimeoutMs',
323
+ type: 'number',
324
+ isOptional: true,
325
+ description:
326
+ 'WebRTC peer disconnect recovery grace period. Defaults to 5000. Failed ICE terminates immediately.',
327
+ },
328
+ {
329
+ name: 'maxPlaybackBufferSeconds',
330
+ type: 'number',
331
+ isOptional: true,
332
+ description:
333
+ 'Continuous WebSocket PCM playback budget before reporting a recoverable gap and pausing playback. Defaults to 2. Unsupported with WebRTC.',
334
+ },
93
335
  {
94
336
  name: 'onToolCall',
95
337
  type: '(options: { toolCall: { toolCallId: string; toolName: string; args: unknown } }) => unknown | Promise<unknown> | undefined',
96
338
  isOptional: true,
97
339
  description:
98
- 'Called when the provider requests a tool call. Return a value to automatically submit it as tool output, or return undefined and call addToolOutput manually later.',
340
+ 'Turn-based tools only. Return a value to automatically submit it as tool output, or return undefined and call addToolOutput manually later. Live client delegations are application-handled through session metadata and events.',
99
341
  },
100
342
  {
101
343
  name: 'onEvent',
@@ -118,14 +360,14 @@ For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
118
360
  content={[
119
361
  {
120
362
  name: 'status',
121
- type: "'disconnected' | 'connecting' | 'connected' | 'error'",
363
+ type: "'disconnected' | 'connecting' | 'connected' | 'closing' | 'error'",
122
364
  description: 'The current connection status.',
123
365
  },
124
366
  {
125
367
  name: 'messages',
126
368
  type: 'UIMessage[]',
127
369
  description:
128
- 'Messages assembled from realtime text, transcript, and tool events.',
370
+ 'Messages assembled from turn-based response text, transcript, and tool events. Continuous Live transcript fragments are exposed separately through session.transcripts.',
129
371
  },
130
372
  {
131
373
  name: 'events',
@@ -136,112 +378,186 @@ For AI Gateway, pass `gateway.experimental_realtime(...)` as the model and point
136
378
  {
137
379
  name: 'isCapturing',
138
380
  type: 'boolean',
139
- description: 'Whether microphone audio capture is active.',
381
+ description:
382
+ 'Capture state observed through SDK controls and events on the selected audio track. External changes to borrowed tracks may require an explicit refresh; see Capture ownership.',
140
383
  },
141
384
  {
142
385
  name: 'isPlaying',
143
386
  type: 'boolean',
144
387
  description: 'Whether model audio playback is active.',
145
388
  },
389
+ {
390
+ name: 'session',
391
+ type: 'Experimental_RealtimeSessionState | undefined',
392
+ description:
393
+ 'Session lifecycle metadata where supported: sessionId, transcript fragments, client delegation metadata, cumulative voice usage, input mute, terminationReason, and pending/confirmed/unconfirmed finalization. Transcript fragments may overlap or arrive late; they are not complete turns.',
394
+ },
146
395
  {
147
396
  name: 'connect',
148
- type: '() => Promise<void>',
397
+ type: '(options?: { stream?: MediaStream; capture?: boolean }) => Promise<void>',
149
398
  description:
150
- 'Fetches the setup token and opens the provider WebSocket connection.',
399
+ 'Starts the selected connection. Continuous browser sessions acquire microphone audio by default; capture: false opts out. A supplied Live stream remains caller-owned. The zero-argument overload remains usable as an event handler.',
400
+ },
401
+ {
402
+ name: 'close',
403
+ type: '(options?: { eventId?: string }) => Promise<void>',
404
+ description:
405
+ 'Stops capture/submissions and waits for bounded graceful finalization where supported. Inspect session.finalization for confirmation; models without acknowledged close disconnect immediately.',
151
406
  },
152
407
  {
153
408
  name: 'disconnect',
154
409
  type: '() => void',
155
- description: 'Closes the provider WebSocket connection.',
410
+ description:
411
+ 'Immediately releases transports and owned media. Preserves latest usage as unconfirmed unless a terminal provider event was received.',
156
412
  },
157
413
  {
158
414
  name: 'addToolOutput',
159
415
  type: '(callId: string, result: unknown) => void',
160
416
  description:
161
- 'Submits the result for a tool call back to the realtime provider.',
417
+ 'Turn-based only: submits a tool result back to the realtime provider. Continuous Live sessions reject this control; append application-handled client delegation results with sendEvent instead.',
162
418
  },
163
419
  {
164
420
  name: 'sendEvent',
165
- type: '(event: Experimental_RealtimeClientEvent) => void',
166
- description: 'Sends a normalized realtime client event.',
421
+ type: '(event: Experimental_RealtimeClientEvent) => Promise<void>',
422
+ description:
423
+ 'Sends a normalized command in order. Resolves on local serialization/submission, not provider acceptance. Validation can throw synchronously; asynchronous send failures reject and are reported through onError. Live context/results use context-append with a known client delegation ID or null.',
167
424
  },
168
425
  {
169
426
  name: 'sendTextMessage',
170
427
  type: '(text: string) => void',
171
- description: 'Sends a user text message and requests a response.',
428
+ description:
429
+ 'Turn-based only: sends user text and requests a response. Continuous Live sessions reject this control. The application owns its text conversation and supplies Live context through sendEvent.',
172
430
  },
173
431
  {
174
432
  name: 'sendAudio',
175
433
  type: '(base64Audio: string) => void',
176
434
  description:
177
- 'Sends a base64-encoded audio chunk to the provider input audio buffer.',
435
+ 'Sends a base64-encoded audio chunk to the provider input audio buffer over WebSocket. WebRTC uses media tracks and rejects JSON audio commands.',
178
436
  },
179
437
  {
180
438
  name: 'commitAudio',
181
439
  type: '() => void',
182
- description: 'Commits the provider input audio buffer.',
440
+ description: 'Turn-based only: commits the provider input audio buffer.',
183
441
  },
184
442
  {
185
443
  name: 'clearAudioBuffer',
186
444
  type: '() => void',
187
- description: 'Clears the provider input audio buffer.',
445
+ description: 'Turn-based only: clears the provider input audio buffer.',
188
446
  },
189
447
  {
190
448
  name: 'requestResponse',
191
449
  type: '(options?: { modalities?: string[] }) => void',
192
- description: 'Requests a new model response.',
450
+ description: 'Turn-based only: requests a new model response.',
193
451
  },
194
452
  {
195
453
  name: 'cancelResponse',
196
454
  type: '() => void',
197
- description: 'Cancels the active model response.',
455
+ description: 'Turn-based only: cancels the active model response.',
198
456
  },
199
457
  {
200
458
  name: 'startAudioCapture',
201
459
  type: '(stream: MediaStream) => void',
202
460
  description:
203
- 'Starts capturing microphone audio from the provided MediaStream.',
461
+ 'Starts or replaces capture with the supplied MediaStream. Live paths borrow tracks; the existing legacy capture API retains ownership of its supplied stream. Async attachment failures are reported through onError.',
204
462
  },
205
463
  {
206
464
  name: 'stopAudioCapture',
207
465
  type: '() => void',
208
- description: 'Stops microphone audio capture.',
466
+ description:
467
+ 'Stops local capture and releases SDK-owned microphone tracks. Caller-owned Live tracks are only detached. Does not change the provider input-mute state.',
468
+ },
469
+ {
470
+ name: 'resumeAudioCapture',
471
+ type: '() => Promise<void>',
472
+ description:
473
+ 'Restarts capture, reusing a supplied Live stream or acquiring a fresh SDK-owned microphone. Does not reconnect the provider.',
209
474
  },
210
475
  {
211
476
  name: 'stopPlayback',
212
477
  type: '() => void',
213
478
  description: 'Stops queued model audio playback.',
214
479
  },
480
+ {
481
+ name: 'resumePlayback',
482
+ type: '() => Promise<void>',
483
+ description:
484
+ 'Retries browser playback after interruption, suspension, or buffer overflow. Discarded stale audio is not replayed.',
485
+ },
215
486
  ]}
216
487
  />
217
488
 
218
489
  ## Tool Calling
219
490
 
220
- Realtime tool execution is client-driven. Use `onToolCall` to handle tool calls
221
- and return the tool output:
491
+ For turn-based models such as `gpt-realtime`, tool execution is client-driven.
492
+ Use `onToolCall` to handle tool calls and return the tool output. Keep the model
493
+ stable across callback updates:
222
494
 
223
495
  ```tsx
224
- const realtime = experimental_useRealtime({
225
- model: openai.experimental_realtime('gpt-realtime'),
226
- api: {
227
- token: '/api/realtime/setup',
228
- },
229
- onToolCall: async ({ toolCall }) => {
230
- if (toolCall.toolName === 'getWeather') {
231
- const response = await fetch('/api/weather', {
232
- method: 'POST',
233
- headers: { 'Content-Type': 'application/json' },
234
- body: JSON.stringify(toolCall.args),
235
- });
236
-
237
- return response.json();
238
- }
239
- },
240
- });
496
+ const model = openai.experimental_realtime('gpt-realtime');
497
+
498
+ function WeatherConversation() {
499
+ const realtime = experimental_useRealtime({
500
+ model,
501
+ api: { token: '/api/realtime/setup' },
502
+ onToolCall: async ({ toolCall }) => {
503
+ if (toolCall.toolName === 'getWeather') {
504
+ const response = await fetch('/api/weather', {
505
+ method: 'POST',
506
+ headers: { 'Content-Type': 'application/json' },
507
+ body: JSON.stringify(toolCall.args),
508
+ });
509
+
510
+ return response.json();
511
+ }
512
+ },
513
+ });
514
+ return <button onClick={realtime.connect}>Connect</button>;
515
+ }
241
516
  ```
242
517
 
243
518
  For tools that require user interaction, return `undefined` from `onToolCall`
244
519
  and call `addToolOutput` later.
245
520
 
521
+ Turn-based providers retain their automatic continuation behavior. Live uses the
522
+ application-handled client delegation flow described above, rather than these tool
523
+ callbacks or turn controls.
524
+
525
+ Outstanding commands are bounded independently from the retained recent-ID history.
526
+ Completed work does not impose a lifetime command quota. Use fresh IDs and keep
527
+ durable operation deduplication in the application; bounded history is not a promise
528
+ of session-long exactly-once execution.
529
+
530
+ ## Capture ownership
531
+
532
+ SDK-acquired tracks are stopped on local capture stop or cleanup. Caller-owned Live
533
+ tracks are detached without stopping or disabling them. `stopAudioCapture()` controls
534
+ local hardware capture; provider `input-audio-mute` controls remote audio processing
535
+ and is tracked through its acknowledgment. These are separate operations: protocol
536
+ mute does not release the microphone, and resuming local capture does not unmute
537
+ provider input. Use `resumeAudioCapture()` to reuse a caller-supplied stream or
538
+ acquire a fresh SDK-owned microphone without reconnecting. The application remains
539
+ responsible for stopping its own tracks when finished with them.
540
+
541
+ `isCapturing` reflects SDK capture controls and events on the selected audio track;
542
+ it does not continuously observe caller-owned media. Assigning `track.enabled`
543
+ does not emit an event, and calling `track.stop()` does not emit an `ended` event.
544
+ After changing borrowed tracks externally, call `startAudioCapture(stream)` or
545
+ `resumeAudioCapture()` to refresh capture state and reattach as needed. If a track
546
+ was stopped, provide a stream with a live audio track; stopped tracks cannot be
547
+ restarted. The SDK does not poll external track state.
548
+
549
+ ## Experimental compatibility
550
+
551
+ This update adds explicit relay options and changes experimental `sendEvent` to
552
+ return a promise. Existing token setups and the no-argument `connect` call remain
553
+ supported. New session state uses `session`, not a provider-branded state object.
554
+ Generic realtime-model consumers must check optional connection methods before calling
555
+ them. Model interfaces and event unions are experimental; exhaustive external switches
556
+ may need to handle the added events. OpenAI uses `experimental_realtime` for both
557
+ Realtime and Live models, with a provider-specific `{ api: 'live' }` override for
558
+ unknown or early-access Live model IDs. This factory option selects the provider API;
559
+ the hook's `api.websocket` option selects the application's transport endpoint.
560
+ OpenAI Live startup options are exported as `Experimental_OpenAIRealtimeModelLiveOptions`.
561
+
246
562
  See [Realtime](/docs/ai-sdk-core/realtime#tool-calling) for a complete example
247
563
  with server-backed app-specific tool endpoints.
@@ -13,7 +13,7 @@ Unlike [`DefaultChatTransport`](/docs/ai-sdk-ui/transport) which assumes the ful
13
13
  'use client';
14
14
 
15
15
  import { useChat } from '@ai-sdk/react';
16
- import { WorkflowChatTransport } from '@ai-sdk/workflow';
16
+ import { WorkflowChatTransport } from '@ai-sdk/workflow/client';
17
17
 
18
18
  export default function Chat() {
19
19
  const { messages, sendMessage } = useChat({
@@ -30,7 +30,7 @@ export default function Chat() {
30
30
  ## Import
31
31
 
32
32
  <Snippet
33
- text={`import { WorkflowChatTransport } from "@ai-sdk/workflow"`}
33
+ text={`import { WorkflowChatTransport } from "@ai-sdk/workflow/client"`}
34
34
  prompt={false}
35
35
  />
36
36
 
@@ -238,7 +238,7 @@ See the [WorkflowAgent guide](/docs/agents/workflow-agent) for complete endpoint
238
238
  'use client';
239
239
 
240
240
  import { useChat } from '@ai-sdk/react';
241
- import { WorkflowChatTransport } from '@ai-sdk/workflow';
241
+ import { WorkflowChatTransport } from '@ai-sdk/workflow/client';
242
242
  import { useMemo } from 'react';
243
243
 
244
244
  export default function Chat() {
@@ -271,7 +271,7 @@ export default function Chat() {
271
271
  'use client';
272
272
 
273
273
  import { useChat } from '@ai-sdk/react';
274
- import { WorkflowChatTransport } from '@ai-sdk/workflow';
274
+ import { WorkflowChatTransport } from '@ai-sdk/workflow/client';
275
275
  import { useMemo } from 'react';
276
276
 
277
277
  export default function Chat() {