@ai-sdk/openai 4.0.66 → 4.0.68
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/README.md +26 -1
- package/dist/index.d.ts +125 -28
- package/dist/index.js +612 -108
- package/dist/index.js.map +1 -1
- package/docs/03-openai.mdx +102 -7
- package/package.json +3 -3
- package/src/index.ts +10 -0
- package/src/live/openai-live-event-mapper.ts +286 -0
- package/src/live/openai-live-session-config.ts +145 -0
- package/src/live/openai-realtime-model-live-options.ts +65 -0
- package/src/live/openai-realtime-model-live.ts +107 -0
- package/src/openai-provider.ts +11 -32
- package/src/realtime/openai-realtime-event-mapper.ts +8 -1
- package/src/realtime/openai-realtime-factory.ts +88 -0
package/docs/03-openai.mdx
CHANGED
|
@@ -39,7 +39,8 @@ You can use the following optional settings to customize the OpenAI provider ins
|
|
|
39
39
|
- **baseURL** _string_
|
|
40
40
|
|
|
41
41
|
Use a different URL prefix for API calls, e.g. to use proxy servers.
|
|
42
|
-
|
|
42
|
+
It defaults to the `OPENAI_BASE_URL` environment variable, and then to
|
|
43
|
+
`https://api.openai.com/v1`.
|
|
43
44
|
The default OpenAI model factory (`openai('model-id')`) uses the Responses
|
|
44
45
|
API. If your custom base URL only supports the Chat Completions API, create
|
|
45
46
|
chat models with `openai.chat('model-id')` instead, or use the
|
|
@@ -2850,17 +2851,38 @@ The following optional provider options are available for OpenAI completion mode
|
|
|
2850
2851
|
|
|
2851
2852
|
<Note type="warning">Realtime is an experimental feature.</Note>
|
|
2852
2853
|
|
|
2853
|
-
|
|
2854
|
-
|
|
2854
|
+
Use `.experimental_realtime()` for both [OpenAI Realtime](https://platform.openai.com/docs/guides/realtime)
|
|
2855
|
+
and [OpenAI Live](https://developers.openai.com/api/docs/guides/live) models.
|
|
2855
2856
|
|
|
2856
2857
|
```ts
|
|
2857
2858
|
import { openai } from '@ai-sdk/openai';
|
|
2858
2859
|
|
|
2859
2860
|
const model = openai.experimental_realtime('gpt-realtime');
|
|
2861
|
+
const liveModel = openai.experimental_realtime('gpt-live-1');
|
|
2862
|
+
|
|
2863
|
+
// Select the API explicitly for an early-access or otherwise unknown model ID.
|
|
2864
|
+
const earlyAccessModel = openai.experimental_realtime('not-yet-released', {
|
|
2865
|
+
api: 'live',
|
|
2866
|
+
});
|
|
2860
2867
|
```
|
|
2861
2868
|
|
|
2862
|
-
|
|
2863
|
-
|
|
2869
|
+
The OpenAI-only `api` option accepts `'live'` or `'realtime'` and takes precedence
|
|
2870
|
+
over model-ID selection. Without it, the known `gpt-live-1` ID uses Live and other
|
|
2871
|
+
IDs retain the Realtime default. The provider does not switch APIs after a failed
|
|
2872
|
+
request. Both model implementations use the existing experimental realtime interface.
|
|
2873
|
+
|
|
2874
|
+
Live supports **browser WSS sessions through an application-owned relay with client
|
|
2875
|
+
delegation**, using `experimental_useRealtime` for the browser runtime, and optional
|
|
2876
|
+
browser-direct WebRTC. Your application owns the agent and tool execution; Live
|
|
2877
|
+
delegations do not invoke `onToolCall`. For a low-level server adapter, obtain server-only connection
|
|
2878
|
+
settings with `liveModel.getServerWebSocketConfig()` and serialize commands with
|
|
2879
|
+
`liveModel.serializeClientEvent()`. Create one parser per connection with
|
|
2880
|
+
`liveModel.createServerEventParser()` and pass incoming JSON to the returned
|
|
2881
|
+
function. It returns arrays of normalized events and retains each original event
|
|
2882
|
+
in `raw`.
|
|
2883
|
+
|
|
2884
|
+
For a token-based **Realtime API** session, create a short-lived token on your
|
|
2885
|
+
server with `openai.experimental_realtime.getToken()`:
|
|
2864
2886
|
|
|
2865
2887
|
```ts
|
|
2866
2888
|
const token = await openai.experimental_realtime.getToken({
|
|
@@ -2868,8 +2890,81 @@ const token = await openai.experimental_realtime.getToken({
|
|
|
2868
2890
|
});
|
|
2869
2891
|
```
|
|
2870
2892
|
|
|
2871
|
-
|
|
2872
|
-
|
|
2893
|
+
Live does not use Realtime client-secret minting. Calling `getToken` for a known
|
|
2894
|
+
Live model or with `api: 'live'` rejects before a mint request. Live's WebSocket
|
|
2895
|
+
connection authenticates on a trusted server; keep the provider API key out of
|
|
2896
|
+
browser code. The API selector chooses a protocol, not a different authentication
|
|
2897
|
+
capability for the same model.
|
|
2898
|
+
|
|
2899
|
+
Browser authentication to an application-owned relay is separate from OpenAI
|
|
2900
|
+
authentication. Protect that relay with your application's session or short-lived
|
|
2901
|
+
credential; an origin check alone is not user authentication. The relay uses its
|
|
2902
|
+
server-held OpenAI key upstream. Selecting a Live model does not make the Realtime
|
|
2903
|
+
client-secret endpoint compatible with the Live API.
|
|
2904
|
+
|
|
2905
|
+
For browser-direct Live WebRTC, an authenticated application endpoint performs the
|
|
2906
|
+
SDP exchange through `liveModel.doCreateWebRTCSession({ sdp, sessionConfig, abortSignal })`
|
|
2907
|
+
using the server's project key. The browser receives the
|
|
2908
|
+
session connection answer, not the project key or a Realtime client secret. This
|
|
2909
|
+
setup endpoint still needs application authentication. See the
|
|
2910
|
+
[`experimental_useRealtime` reference](/docs/reference/ai-sdk-ui/use-realtime)
|
|
2911
|
+
for the `api.session` setup contract and the runnable example.
|
|
2912
|
+
|
|
2913
|
+
### Live session options
|
|
2914
|
+
|
|
2915
|
+
Configure Live-specific settings through `sessionConfig.providerOptions.openai`,
|
|
2916
|
+
typed with `Experimental_OpenAIRealtimeModelLiveOptions`. For example:
|
|
2917
|
+
|
|
2918
|
+
```ts
|
|
2919
|
+
import type { Experimental_OpenAIRealtimeModelLiveOptions as OpenAIRealtimeModelLiveOptions } from '@ai-sdk/openai';
|
|
2920
|
+
|
|
2921
|
+
const sessionConfig = {
|
|
2922
|
+
instructions: 'Be concise. Delegate questions that need research or tools.',
|
|
2923
|
+
providerOptions: {
|
|
2924
|
+
openai: {
|
|
2925
|
+
delegation: { type: 'client' },
|
|
2926
|
+
} satisfies OpenAIRealtimeModelLiveOptions,
|
|
2927
|
+
},
|
|
2928
|
+
};
|
|
2929
|
+
```
|
|
2930
|
+
|
|
2931
|
+
- `delegation`: use `{ type: 'client' }` for application-owned work. Omission or
|
|
2932
|
+
`null` also selects client mode. Other delegation modes are unsupported.
|
|
2933
|
+
- `input`: prior text messages with developer, user or assistant roles.
|
|
2934
|
+
- `voice`: an authorized custom voice reference `{ id }`, mutually exclusive with
|
|
2935
|
+
the ordinary `sessionConfig.voice` string.
|
|
2936
|
+
- `store`: requests provider-side session storage; availability depends on the
|
|
2937
|
+
OpenAI project configuration.
|
|
2938
|
+
- `client.dataChannel`: WebRTC startup permission policy with
|
|
2939
|
+
`allowedClientEvents` and `allowedServerEvents`. Set it on the application server,
|
|
2940
|
+
not from trusted-looking browser input. Omission retains the provider default;
|
|
2941
|
+
`[]` allows none and `'all'` allows all. Response event selectors use
|
|
2942
|
+
`{ type: 'response.event', responseEvent: 'response.completed' }` to select
|
|
2943
|
+
native nested events, which remain raw `custom` events rather than managed SDK
|
|
2944
|
+
responses. Keep readiness, errors, and terminal usage observable for
|
|
2945
|
+
the session features your application enables.
|
|
2946
|
+
|
|
2947
|
+
Send the initial WebSocket configuration with `session-start`; WebRTC starts the
|
|
2948
|
+
session during SDP setup and negotiates audio formats through SDP. Live startup settings are
|
|
2949
|
+
immutable: `session-update` is rejected. Use `input-audio-mute` and
|
|
2950
|
+
`input-audio-unmute` for remote input processing; these commands do not release the
|
|
2951
|
+
local microphone. Return application context through
|
|
2952
|
+
`context-append`, using the opaque `delegationId` from `delegation-created`, or
|
|
2953
|
+
`null` for context without a delegation. Its `providerOptions.openai.channel`
|
|
2954
|
+
accepts `instructions`, `thinking` (the default), or `commentary`.
|
|
2955
|
+
|
|
2956
|
+
For graceful shutdown, install a `session-closed` listener before sending `session-close`;
|
|
2957
|
+
that event confirms final voice usage. `session-usage` contains cumulative seconds,
|
|
2958
|
+
so replace earlier snapshots rather than adding them.
|
|
2959
|
+
|
|
2960
|
+
For the browser integration, see the
|
|
2961
|
+
[`experimental_useRealtime` reference](/docs/reference/ai-sdk-ui/use-realtime#continuous-conversations).
|
|
2962
|
+
It covers the JSON/PCM16 WebSocket relay runtime, optional WebRTC setup and permission
|
|
2963
|
+
policy, microphone and playback ownership, `session.delegations`, `sendEvent` for
|
|
2964
|
+
application context, and `close()` for bounded finalization. Use `wss:` for production
|
|
2965
|
+
relays. Continuous conversation semantics do not imply browser support for every
|
|
2966
|
+
provider codec or transport. The [Realtime guide](/docs/ai-sdk-core/realtime) covers
|
|
2967
|
+
legacy token-based, turn-based sessions.
|
|
2873
2968
|
|
|
2874
2969
|
## Embedding Models
|
|
2875
2970
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/openai",
|
|
3
|
-
"version": "4.0.
|
|
3
|
+
"version": "4.0.68",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"sideEffects": false,
|
|
@@ -35,8 +35,8 @@
|
|
|
35
35
|
}
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@ai-sdk/provider": "4.0.
|
|
39
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
38
|
+
"@ai-sdk/provider": "4.0.16",
|
|
39
|
+
"@ai-sdk/provider-utils": "5.0.42"
|
|
40
40
|
},
|
|
41
41
|
"devDependencies": {
|
|
42
42
|
"@ai-sdk/test-server": "2.0.1",
|
package/src/index.ts
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
export { createOpenAI, openai } from './openai-provider';
|
|
2
2
|
export type { OpenAIProvider, OpenAIProviderSettings } from './openai-provider';
|
|
3
|
+
export type {
|
|
4
|
+
OpenAIRealtimeFactory as Experimental_OpenAIRealtimeFactory,
|
|
5
|
+
OpenAIRealtimeOptions as Experimental_OpenAIRealtimeOptions,
|
|
6
|
+
} from './realtime/openai-realtime-factory';
|
|
7
|
+
export { OpenAIRealtimeModelLive as Experimental_OpenAIRealtimeModelLive } from './live/openai-realtime-model-live';
|
|
8
|
+
export type { OpenAIRealtimeModelLiveConfig as Experimental_OpenAIRealtimeModelLiveConfig } from './live/openai-realtime-model-live';
|
|
9
|
+
export type {
|
|
10
|
+
OpenAIRealtimeModelLiveId as Experimental_OpenAIRealtimeModelLiveId,
|
|
11
|
+
OpenAIRealtimeModelLiveOptions as Experimental_OpenAIRealtimeModelLiveOptions,
|
|
12
|
+
} from './live/openai-realtime-model-live-options';
|
|
3
13
|
export { OpenAIRealtimeModel as Experimental_OpenAIRealtimeModel } from './realtime/openai-realtime-model';
|
|
4
14
|
export type { OpenAIRealtimeModelConfig as Experimental_OpenAIRealtimeModelConfig } from './realtime/openai-realtime-model';
|
|
5
15
|
export type {
|
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
import {
|
|
2
|
+
UnsupportedFunctionalityError,
|
|
3
|
+
type Experimental_RealtimeModelV4ClientEvent as RealtimeModelV4ClientEvent,
|
|
4
|
+
type Experimental_RealtimeModelV4ServerEvent as RealtimeModelV4ServerEvent,
|
|
5
|
+
} from '@ai-sdk/provider';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
import { buildOpenAILiveSessionConfig } from './openai-live-session-config';
|
|
8
|
+
|
|
9
|
+
const sessionSchema = z.object({ id: z.string().min(1) });
|
|
10
|
+
const startedSessionSchema = sessionSchema.extend({
|
|
11
|
+
delegation: z.object({ type: z.enum(['client', 'responses']) }).nullish(),
|
|
12
|
+
});
|
|
13
|
+
const usageSchema = z.object({ seconds: z.number().nonnegative() });
|
|
14
|
+
const transcriptFields = {
|
|
15
|
+
delta: z.string(),
|
|
16
|
+
start_ms: z.number().nonnegative(),
|
|
17
|
+
end_ms: z.number().nonnegative(),
|
|
18
|
+
};
|
|
19
|
+
const acknowledgmentFields = { client_event_id: z.string().nullish() };
|
|
20
|
+
const appendAcknowledgmentFields = {
|
|
21
|
+
...acknowledgmentFields,
|
|
22
|
+
start_ms: z.number().nonnegative(),
|
|
23
|
+
end_ms: z.number().nonnegative(),
|
|
24
|
+
};
|
|
25
|
+
const serverEventSchema = z.discriminatedUnion('type', [
|
|
26
|
+
z.object({
|
|
27
|
+
type: z.literal('session.started'),
|
|
28
|
+
session: startedSessionSchema,
|
|
29
|
+
}),
|
|
30
|
+
z.object({
|
|
31
|
+
type: z.literal('session.closed'),
|
|
32
|
+
session: sessionSchema.nullish(),
|
|
33
|
+
usage: usageSchema,
|
|
34
|
+
reason: z.string(),
|
|
35
|
+
}),
|
|
36
|
+
z.object({
|
|
37
|
+
type: z.literal('session.usage.updated'),
|
|
38
|
+
usage: usageSchema,
|
|
39
|
+
context_window: z
|
|
40
|
+
.object({ usage_ratio: z.number().min(0).max(1) })
|
|
41
|
+
.nullish(),
|
|
42
|
+
}),
|
|
43
|
+
z.object({
|
|
44
|
+
type: z.literal('session.output_audio.delta'),
|
|
45
|
+
delta: z.string(),
|
|
46
|
+
}),
|
|
47
|
+
z.object({
|
|
48
|
+
type: z.literal('session.input_transcript.delta'),
|
|
49
|
+
...transcriptFields,
|
|
50
|
+
}),
|
|
51
|
+
z.object({
|
|
52
|
+
type: z.literal('session.output_transcript.delta'),
|
|
53
|
+
...transcriptFields,
|
|
54
|
+
}),
|
|
55
|
+
z.object({
|
|
56
|
+
type: z.literal('session.delegation.created'),
|
|
57
|
+
offset_ms: z.number().nonnegative().nullish(),
|
|
58
|
+
delegation: z.object({
|
|
59
|
+
id: z.string().min(1),
|
|
60
|
+
target: z.enum(['client', 'responses']).nullish(),
|
|
61
|
+
response_id: z.string().min(1).nullish(),
|
|
62
|
+
}),
|
|
63
|
+
}),
|
|
64
|
+
z.object({
|
|
65
|
+
type: z.literal('session.updated'),
|
|
66
|
+
session: sessionSchema,
|
|
67
|
+
...acknowledgmentFields,
|
|
68
|
+
}),
|
|
69
|
+
z.object({
|
|
70
|
+
type: z.literal('session.input_audio.muted'),
|
|
71
|
+
...acknowledgmentFields,
|
|
72
|
+
}),
|
|
73
|
+
z.object({
|
|
74
|
+
type: z.literal('session.input_audio.unmuted'),
|
|
75
|
+
...acknowledgmentFields,
|
|
76
|
+
}),
|
|
77
|
+
z.object({
|
|
78
|
+
type: z.literal('session.instructions.appended'),
|
|
79
|
+
...appendAcknowledgmentFields,
|
|
80
|
+
}),
|
|
81
|
+
z.object({
|
|
82
|
+
type: z.literal('session.thinking.appended'),
|
|
83
|
+
...appendAcknowledgmentFields,
|
|
84
|
+
}),
|
|
85
|
+
z.object({
|
|
86
|
+
type: z.literal('session.commentary.appended'),
|
|
87
|
+
...appendAcknowledgmentFields,
|
|
88
|
+
}),
|
|
89
|
+
z.object({
|
|
90
|
+
type: z.literal('error'),
|
|
91
|
+
error: z.object({
|
|
92
|
+
message: z.string(),
|
|
93
|
+
code: z.string().nullish(),
|
|
94
|
+
client_event_id: z.string().nullish(),
|
|
95
|
+
}),
|
|
96
|
+
}),
|
|
97
|
+
]);
|
|
98
|
+
const knownTypes = new Set<string>(
|
|
99
|
+
serverEventSchema.options.map(schema => schema.shape.type.value),
|
|
100
|
+
);
|
|
101
|
+
const envelopeSchema = z.object({ type: z.string() });
|
|
102
|
+
|
|
103
|
+
export function createOpenAILiveServerEventParser(): (
|
|
104
|
+
raw: unknown,
|
|
105
|
+
) => RealtimeModelV4ServerEvent[] {
|
|
106
|
+
return raw => parseOpenAILiveServerEvent(raw);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
export function parseOpenAILiveServerEvent(
|
|
110
|
+
raw: unknown,
|
|
111
|
+
): RealtimeModelV4ServerEvent[] {
|
|
112
|
+
return [parseServerEvent(raw)];
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
function parseServerEvent(raw: unknown): RealtimeModelV4ServerEvent {
|
|
116
|
+
const envelope = envelopeSchema.safeParse(raw);
|
|
117
|
+
if (envelope.success && !knownTypes.has(envelope.data.type)) {
|
|
118
|
+
return { type: 'custom', rawType: envelope.data.type, raw };
|
|
119
|
+
}
|
|
120
|
+
const parsed = serverEventSchema.safeParse(raw);
|
|
121
|
+
if (!parsed.success) {
|
|
122
|
+
return {
|
|
123
|
+
type: 'error',
|
|
124
|
+
code: 'invalid_server_event',
|
|
125
|
+
message: 'Invalid OpenAI Live server event.',
|
|
126
|
+
raw,
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
const event = parsed.data;
|
|
130
|
+
if ('start_ms' in event && event.end_ms < event.start_ms) {
|
|
131
|
+
return {
|
|
132
|
+
type: 'error',
|
|
133
|
+
code: 'invalid_server_event',
|
|
134
|
+
message: 'Invalid OpenAI Live event time interval.',
|
|
135
|
+
raw,
|
|
136
|
+
};
|
|
137
|
+
}
|
|
138
|
+
switch (event.type) {
|
|
139
|
+
case 'session.started':
|
|
140
|
+
return {
|
|
141
|
+
type: 'session-started',
|
|
142
|
+
sessionId: event.session.id,
|
|
143
|
+
delegationMode:
|
|
144
|
+
event.session.delegation?.type === 'responses'
|
|
145
|
+
? 'provider'
|
|
146
|
+
: 'client',
|
|
147
|
+
raw,
|
|
148
|
+
};
|
|
149
|
+
case 'session.closed':
|
|
150
|
+
return {
|
|
151
|
+
type: 'session-closed',
|
|
152
|
+
sessionId: event.session?.id,
|
|
153
|
+
usage: event.usage,
|
|
154
|
+
reason: event.reason,
|
|
155
|
+
raw,
|
|
156
|
+
};
|
|
157
|
+
case 'session.usage.updated':
|
|
158
|
+
return {
|
|
159
|
+
type: 'session-usage',
|
|
160
|
+
usage: event.usage,
|
|
161
|
+
contextWindowUsageRatio: event.context_window?.usage_ratio,
|
|
162
|
+
raw,
|
|
163
|
+
};
|
|
164
|
+
case 'session.output_audio.delta':
|
|
165
|
+
return { type: 'audio-chunk', delta: event.delta, raw };
|
|
166
|
+
case 'session.input_transcript.delta':
|
|
167
|
+
case 'session.output_transcript.delta':
|
|
168
|
+
return {
|
|
169
|
+
type: 'transcript-fragment',
|
|
170
|
+
speaker:
|
|
171
|
+
event.type === 'session.input_transcript.delta'
|
|
172
|
+
? 'user'
|
|
173
|
+
: 'assistant',
|
|
174
|
+
delta: event.delta,
|
|
175
|
+
startMs: event.start_ms,
|
|
176
|
+
endMs: event.end_ms,
|
|
177
|
+
raw,
|
|
178
|
+
};
|
|
179
|
+
case 'session.delegation.created':
|
|
180
|
+
return {
|
|
181
|
+
type: 'delegation-created',
|
|
182
|
+
delegationId: event.delegation.id,
|
|
183
|
+
target:
|
|
184
|
+
event.delegation.target === 'responses'
|
|
185
|
+
? 'provider'
|
|
186
|
+
: (event.delegation.target ?? undefined),
|
|
187
|
+
offsetMs: event.offset_ms ?? undefined,
|
|
188
|
+
...(event.delegation.response_id != null
|
|
189
|
+
? { responseId: event.delegation.response_id }
|
|
190
|
+
: {}),
|
|
191
|
+
raw,
|
|
192
|
+
};
|
|
193
|
+
case 'error':
|
|
194
|
+
return {
|
|
195
|
+
type: 'error',
|
|
196
|
+
message: event.error.message,
|
|
197
|
+
code: event.error.code ?? undefined,
|
|
198
|
+
clientEventId: event.error.client_event_id ?? undefined,
|
|
199
|
+
raw,
|
|
200
|
+
};
|
|
201
|
+
case 'session.updated':
|
|
202
|
+
return {
|
|
203
|
+
type: 'command-acknowledged',
|
|
204
|
+
command: 'session.update',
|
|
205
|
+
clientEventId: event.client_event_id ?? undefined,
|
|
206
|
+
raw,
|
|
207
|
+
};
|
|
208
|
+
case 'session.input_audio.muted':
|
|
209
|
+
case 'session.input_audio.unmuted':
|
|
210
|
+
return {
|
|
211
|
+
type: 'command-acknowledged',
|
|
212
|
+
command: event.type.slice(0, -1),
|
|
213
|
+
clientEventId: event.client_event_id ?? undefined,
|
|
214
|
+
raw,
|
|
215
|
+
};
|
|
216
|
+
case 'session.instructions.appended':
|
|
217
|
+
case 'session.thinking.appended':
|
|
218
|
+
case 'session.commentary.appended':
|
|
219
|
+
return {
|
|
220
|
+
type: 'command-acknowledged',
|
|
221
|
+
command: event.type.slice(0, -2),
|
|
222
|
+
clientEventId: event.client_event_id ?? undefined,
|
|
223
|
+
raw,
|
|
224
|
+
};
|
|
225
|
+
}
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
export function serializeOpenAILiveClientEvent(
|
|
229
|
+
event: RealtimeModelV4ClientEvent,
|
|
230
|
+
modelId: string,
|
|
231
|
+
): unknown {
|
|
232
|
+
const eventId =
|
|
233
|
+
'eventId' in event && event.eventId !== undefined
|
|
234
|
+
? { event_id: event.eventId }
|
|
235
|
+
: {};
|
|
236
|
+
switch (event.type) {
|
|
237
|
+
case 'session-start':
|
|
238
|
+
return {
|
|
239
|
+
type: 'session.start',
|
|
240
|
+
session: buildOpenAILiveSessionConfig(event.config, modelId),
|
|
241
|
+
...eventId,
|
|
242
|
+
};
|
|
243
|
+
case 'session-update':
|
|
244
|
+
throw new UnsupportedFunctionalityError({
|
|
245
|
+
functionality:
|
|
246
|
+
'OpenAI Live session-update; startup settings are immutable; use context-append or input-audio-mute/input-audio-unmute',
|
|
247
|
+
});
|
|
248
|
+
case 'session-close':
|
|
249
|
+
return { type: 'session.close', ...eventId };
|
|
250
|
+
case 'input-audio-append':
|
|
251
|
+
return {
|
|
252
|
+
type: 'session.input_audio.append',
|
|
253
|
+
audio: event.audio,
|
|
254
|
+
...eventId,
|
|
255
|
+
};
|
|
256
|
+
case 'input-audio-mute':
|
|
257
|
+
return { type: 'session.input_audio.mute', ...eventId };
|
|
258
|
+
case 'input-audio-unmute':
|
|
259
|
+
return { type: 'session.input_audio.unmute', ...eventId };
|
|
260
|
+
case 'context-append': {
|
|
261
|
+
const context = z
|
|
262
|
+
.object({
|
|
263
|
+
content: z.string(),
|
|
264
|
+
delegationId: z.string().min(1).nullable(),
|
|
265
|
+
})
|
|
266
|
+
.parse(event);
|
|
267
|
+
const options = z
|
|
268
|
+
.strictObject({
|
|
269
|
+
channel: z
|
|
270
|
+
.enum(['instructions', 'thinking', 'commentary'])
|
|
271
|
+
.optional(),
|
|
272
|
+
})
|
|
273
|
+
.parse(event.providerOptions?.openai ?? {});
|
|
274
|
+
return {
|
|
275
|
+
type: `session.${options.channel ?? 'thinking'}.append`,
|
|
276
|
+
content: context.content,
|
|
277
|
+
delegation_id: context.delegationId,
|
|
278
|
+
...eventId,
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
default:
|
|
282
|
+
throw new UnsupportedFunctionalityError({
|
|
283
|
+
functionality: `OpenAI Live command: ${event.type}; use continuous audio and context-append instead of voice-turn commands`,
|
|
284
|
+
});
|
|
285
|
+
}
|
|
286
|
+
}
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
import {
|
|
2
|
+
InvalidArgumentError,
|
|
3
|
+
UnsupportedFunctionalityError,
|
|
4
|
+
type Experimental_RealtimeModelV4SessionConfig as RealtimeModelV4SessionConfig,
|
|
5
|
+
} from '@ai-sdk/provider';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
import { openaiRealtimeModelLiveOptionsSchema } from './openai-realtime-model-live-options';
|
|
8
|
+
|
|
9
|
+
const audioFormatSchema = z.union([
|
|
10
|
+
z.strictObject({
|
|
11
|
+
type: z.literal('audio/pcm'),
|
|
12
|
+
rate: z.union([z.literal(16000), z.literal(24000)]),
|
|
13
|
+
}),
|
|
14
|
+
z.strictObject({
|
|
15
|
+
type: z.enum(['audio/pcma', 'audio/pcmu']),
|
|
16
|
+
rate: z.literal(8000),
|
|
17
|
+
}),
|
|
18
|
+
]);
|
|
19
|
+
|
|
20
|
+
export function buildOpenAILiveSessionConfig(
|
|
21
|
+
config: RealtimeModelV4SessionConfig,
|
|
22
|
+
modelId: string,
|
|
23
|
+
transport: 'websocket' | 'webrtc' = 'websocket',
|
|
24
|
+
): Record<string, unknown> {
|
|
25
|
+
for (const key of Object.keys(config)) {
|
|
26
|
+
if (
|
|
27
|
+
![
|
|
28
|
+
'instructions',
|
|
29
|
+
'voice',
|
|
30
|
+
'inputAudioFormat',
|
|
31
|
+
'outputAudioFormat',
|
|
32
|
+
'providerOptions',
|
|
33
|
+
].includes(key)
|
|
34
|
+
) {
|
|
35
|
+
throw new UnsupportedFunctionalityError({
|
|
36
|
+
functionality: `OpenAI Live session setting: ${key}`,
|
|
37
|
+
});
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
if (
|
|
42
|
+
z
|
|
43
|
+
.object({ delegation: z.object({ type: z.literal('responses') }) })
|
|
44
|
+
.safeParse(config.providerOptions?.openai).success
|
|
45
|
+
) {
|
|
46
|
+
throw new UnsupportedFunctionalityError({
|
|
47
|
+
functionality:
|
|
48
|
+
'OpenAI Live Responses delegation; only client delegation is supported',
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
const options = openaiRealtimeModelLiveOptionsSchema.parse(
|
|
52
|
+
config.providerOptions?.openai ?? {},
|
|
53
|
+
);
|
|
54
|
+
if (options.client !== undefined && transport !== 'webrtc') {
|
|
55
|
+
throw new UnsupportedFunctionalityError({
|
|
56
|
+
functionality: 'OpenAI Live client permissions outside WebRTC startup',
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
if (options.voice != null && config.voice != null) {
|
|
60
|
+
throw new InvalidArgumentError({
|
|
61
|
+
argument: 'voice',
|
|
62
|
+
message: 'Choose either voice or providerOptions.openai.voice.',
|
|
63
|
+
});
|
|
64
|
+
}
|
|
65
|
+
if (
|
|
66
|
+
transport === 'webrtc' &&
|
|
67
|
+
(config.inputAudioFormat !== undefined ||
|
|
68
|
+
config.outputAudioFormat !== undefined)
|
|
69
|
+
) {
|
|
70
|
+
throw new UnsupportedFunctionalityError({
|
|
71
|
+
functionality:
|
|
72
|
+
'Fixed audio formats for OpenAI Live WebRTC; audio is negotiated through SDP',
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const inputFormat =
|
|
77
|
+
config.inputAudioFormat == null
|
|
78
|
+
? undefined
|
|
79
|
+
: audioFormatSchema.parse(config.inputAudioFormat);
|
|
80
|
+
const outputFormat =
|
|
81
|
+
config.outputAudioFormat == null
|
|
82
|
+
? undefined
|
|
83
|
+
: audioFormatSchema.parse(config.outputAudioFormat);
|
|
84
|
+
if (
|
|
85
|
+
inputFormat != null &&
|
|
86
|
+
outputFormat != null &&
|
|
87
|
+
(inputFormat.type !== outputFormat.type ||
|
|
88
|
+
inputFormat.rate !== outputFormat.rate)
|
|
89
|
+
) {
|
|
90
|
+
throw new InvalidArgumentError({
|
|
91
|
+
argument: 'outputAudioFormat',
|
|
92
|
+
message: 'OpenAI Live requires the same input and output audio format.',
|
|
93
|
+
});
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
return {
|
|
97
|
+
model: modelId,
|
|
98
|
+
...(config.instructions !== undefined
|
|
99
|
+
? { instructions: config.instructions }
|
|
100
|
+
: {}),
|
|
101
|
+
audio: {
|
|
102
|
+
...(transport === 'websocket'
|
|
103
|
+
? {
|
|
104
|
+
format: inputFormat ??
|
|
105
|
+
outputFormat ?? { type: 'audio/pcm', rate: 24000 },
|
|
106
|
+
}
|
|
107
|
+
: {}),
|
|
108
|
+
output: { voice: options.voice ?? config.voice ?? 'marin' },
|
|
109
|
+
},
|
|
110
|
+
...(options.delegation !== undefined
|
|
111
|
+
? { delegation: options.delegation }
|
|
112
|
+
: {}),
|
|
113
|
+
...(options.client !== undefined
|
|
114
|
+
? {
|
|
115
|
+
client: {
|
|
116
|
+
data_channel: {
|
|
117
|
+
...(options.client.dataChannel.allowedClientEvents !== undefined
|
|
118
|
+
? {
|
|
119
|
+
allowed_client_events:
|
|
120
|
+
options.client.dataChannel.allowedClientEvents,
|
|
121
|
+
}
|
|
122
|
+
: {}),
|
|
123
|
+
...(options.client.dataChannel.allowedServerEvents !== undefined
|
|
124
|
+
? {
|
|
125
|
+
allowed_server_events:
|
|
126
|
+
options.client.dataChannel.allowedServerEvents === 'all'
|
|
127
|
+
? 'all'
|
|
128
|
+
: options.client.dataChannel.allowedServerEvents.map(
|
|
129
|
+
selector => ({
|
|
130
|
+
type: selector.type,
|
|
131
|
+
...(selector.responseEvent !== undefined
|
|
132
|
+
? { response_event: selector.responseEvent }
|
|
133
|
+
: {}),
|
|
134
|
+
}),
|
|
135
|
+
),
|
|
136
|
+
}
|
|
137
|
+
: {}),
|
|
138
|
+
},
|
|
139
|
+
},
|
|
140
|
+
}
|
|
141
|
+
: {}),
|
|
142
|
+
...(options.input !== undefined ? { input: options.input } : {}),
|
|
143
|
+
...(options.store !== undefined ? { store: options.store } : {}),
|
|
144
|
+
};
|
|
145
|
+
}
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import { z } from 'zod/v4';
|
|
2
|
+
|
|
3
|
+
export type OpenAIRealtimeModelLiveId = 'gpt-live-1' | (string & {});
|
|
4
|
+
|
|
5
|
+
const serverEventSelectorSchema = z
|
|
6
|
+
.strictObject({
|
|
7
|
+
type: z.string(),
|
|
8
|
+
responseEvent: z.string().optional(),
|
|
9
|
+
})
|
|
10
|
+
.refine(
|
|
11
|
+
selector =>
|
|
12
|
+
(selector.type === 'response.event') ===
|
|
13
|
+
(selector.responseEvent !== undefined),
|
|
14
|
+
'responseEvent is required for response.event and forbidden for other event types.',
|
|
15
|
+
);
|
|
16
|
+
|
|
17
|
+
export const openaiRealtimeModelLiveOptionsSchema = z.strictObject({
|
|
18
|
+
client: z
|
|
19
|
+
.strictObject({
|
|
20
|
+
dataChannel: z.strictObject({
|
|
21
|
+
allowedClientEvents: z
|
|
22
|
+
.union([z.literal('all'), z.array(z.string())])
|
|
23
|
+
.optional(),
|
|
24
|
+
allowedServerEvents: z
|
|
25
|
+
.union([z.literal('all'), z.array(serverEventSelectorSchema)])
|
|
26
|
+
.optional(),
|
|
27
|
+
}),
|
|
28
|
+
})
|
|
29
|
+
.optional(),
|
|
30
|
+
delegation: z
|
|
31
|
+
.strictObject({ type: z.literal('client') })
|
|
32
|
+
.nullable()
|
|
33
|
+
.optional(),
|
|
34
|
+
input: z
|
|
35
|
+
.array(
|
|
36
|
+
z.discriminatedUnion('role', [
|
|
37
|
+
z.strictObject({
|
|
38
|
+
type: z.literal('message'),
|
|
39
|
+
role: z.enum(['developer', 'user']),
|
|
40
|
+
content: z.tuple([
|
|
41
|
+
z.strictObject({ type: z.literal('input_text'), text: z.string() }),
|
|
42
|
+
]),
|
|
43
|
+
}),
|
|
44
|
+
z.strictObject({
|
|
45
|
+
type: z.literal('message'),
|
|
46
|
+
role: z.literal('assistant'),
|
|
47
|
+
content: z.tuple([
|
|
48
|
+
z.strictObject({
|
|
49
|
+
type: z.enum(['text', 'output_text']),
|
|
50
|
+
text: z.string(),
|
|
51
|
+
}),
|
|
52
|
+
]),
|
|
53
|
+
}),
|
|
54
|
+
]),
|
|
55
|
+
)
|
|
56
|
+
.max(128)
|
|
57
|
+
.optional(),
|
|
58
|
+
store: z.boolean().optional(),
|
|
59
|
+
voice: z.strictObject({ id: z.string().min(1) }).optional(),
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
/** Experimental Live options under sessionConfig.providerOptions.openai. */
|
|
63
|
+
export type OpenAIRealtimeModelLiveOptions = z.infer<
|
|
64
|
+
typeof openaiRealtimeModelLiveOptionsSchema
|
|
65
|
+
>;
|