@ai-sdk/openai 4.0.21 → 4.0.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/dist/index.d.ts +65 -3
- package/dist/index.js +357 -24
- package/dist/index.js.map +1 -1
- package/dist/internal/index.d.ts +8 -0
- package/dist/internal/index.js +8 -2
- package/dist/internal/index.js.map +1 -1
- package/docs/03-openai.mdx +49 -1
- package/package.json +2 -2
- package/src/index.ts +5 -0
- package/src/openai-provider.ts +27 -0
- package/src/responses/openai-responses-api.ts +6 -1
- package/src/responses/openai-responses-prepare-tools.ts +4 -1
- package/src/tool/web-search.ts +12 -1
- package/src/translation/openai-translation-model-options.ts +16 -0
- package/src/translation/openai-translation-model.ts +397 -0
package/docs/03-openai.mdx
CHANGED
|
@@ -539,6 +539,7 @@ const result = await generateText({
|
|
|
539
539
|
},
|
|
540
540
|
filters: {
|
|
541
541
|
allowedDomains: ['sfchronicle.com', 'sfgate.com'],
|
|
542
|
+
blockedDomains: ['example.com'],
|
|
542
543
|
},
|
|
543
544
|
}),
|
|
544
545
|
},
|
|
@@ -565,7 +566,10 @@ The web search tool supports the following configuration options:
|
|
|
565
566
|
- **searchContextSize** _'low' | 'medium' | 'high'_ - Controls the amount of context used for the search. Higher values provide more comprehensive results but may have higher latency and cost.
|
|
566
567
|
- **userLocation** - Optional location information to provide geographically relevant results. Includes `type` (always `'approximate'`), `country`, `city`, `region`, and `timezone`.
|
|
567
568
|
- **filters** - Optional filter configuration to restrict search results.
|
|
568
|
-
- **allowedDomains** _string[]_ -
|
|
569
|
+
- **allowedDomains** _string[]_ - Up to 100 allowed domains for the search.
|
|
570
|
+
- **blockedDomains** _string[]_ - Up to 100 blocked domains for the search.
|
|
571
|
+
|
|
572
|
+
Omit the HTTP or HTTPS prefix from domain filters. Subdomains of configured domains are automatically included or excluded.
|
|
569
573
|
|
|
570
574
|
For detailed information on configuration options see the [OpenAI Web Search Tool documentation](https://platform.openai.com/docs/guides/tools-web-search?api-mode=responses).
|
|
571
575
|
|
|
@@ -2903,6 +2907,50 @@ The following provider options are available:
|
|
|
2903
2907
|
| `gpt-4o-transcribe` | <Check /> | <Cross /> | <Cross /> | <Cross /> | <Cross /> |
|
|
2904
2908
|
| `gpt-realtime-whisper` | <Cross /> | <Check /> | <Cross /> | <Cross /> | <Cross /> |
|
|
2905
2909
|
|
|
2910
|
+
## Translation Models
|
|
2911
|
+
|
|
2912
|
+
<Note type="warning">Speech translation is an experimental feature.</Note>
|
|
2913
|
+
|
|
2914
|
+
You can create models that translate live speech over the OpenAI Realtime
|
|
2915
|
+
translations WebSocket using the `.translation()` factory method. Translation
|
|
2916
|
+
models are streaming-only and are used with
|
|
2917
|
+
[`experimental_streamTranslate`](/docs/reference/ai-sdk-core/stream-translate).
|
|
2918
|
+
|
|
2919
|
+
The first argument is the model id e.g. `gpt-realtime-translate`.
|
|
2920
|
+
|
|
2921
|
+
```ts
|
|
2922
|
+
const model = openai.translation('gpt-realtime-translate');
|
|
2923
|
+
```
|
|
2924
|
+
|
|
2925
|
+
```ts
|
|
2926
|
+
import { experimental_streamTranslate as streamTranslate } from 'ai';
|
|
2927
|
+
import { openai } from '@ai-sdk/openai';
|
|
2928
|
+
|
|
2929
|
+
const result = streamTranslate({
|
|
2930
|
+
model: openai.translation('gpt-realtime-translate'),
|
|
2931
|
+
audio: audioStream, // ReadableStream<Uint8Array | string>
|
|
2932
|
+
inputAudioFormat: { type: 'audio/pcm', rate: 24000 },
|
|
2933
|
+
targetLanguage: 'es',
|
|
2934
|
+
});
|
|
2935
|
+
|
|
2936
|
+
for await (const part of result.fullStream) {
|
|
2937
|
+
if (part.type === 'output-text-delta') {
|
|
2938
|
+
process.stdout.write(part.delta);
|
|
2939
|
+
}
|
|
2940
|
+
}
|
|
2941
|
+
```
|
|
2942
|
+
|
|
2943
|
+
The OpenAI Realtime translation API auto-detects the source language
|
|
2944
|
+
(`sourceLanguage` is not supported), accepts 24kHz 16-bit PCM input, and always
|
|
2945
|
+
outputs 24kHz 16-bit PCM audio (`outputAudioFormat` is not supported).
|
|
2946
|
+
Unsupported source and output settings surface as call warnings.
|
|
2947
|
+
|
|
2948
|
+
### Model Capabilities
|
|
2949
|
+
|
|
2950
|
+
| Model | Translated Audio | Translated Text | Source Transcript |
|
|
2951
|
+
| ------------------------ | ---------------- | --------------- | ----------------- |
|
|
2952
|
+
| `gpt-realtime-translate` | <Check /> | <Check /> | <Check /> |
|
|
2953
|
+
|
|
2906
2954
|
## Speech Models
|
|
2907
2955
|
|
|
2908
2956
|
You can create models that call the [OpenAI speech API](https://platform.openai.com/docs/api-reference/audio/speech)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ai-sdk/openai",
|
|
3
|
-
"version": "4.0.
|
|
3
|
+
"version": "4.0.23",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"sideEffects": false,
|
|
@@ -36,7 +36,7 @@
|
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
38
|
"@ai-sdk/provider": "4.0.4",
|
|
39
|
-
"@ai-sdk/provider-utils": "5.0.
|
|
39
|
+
"@ai-sdk/provider-utils": "5.0.14"
|
|
40
40
|
},
|
|
41
41
|
"devDependencies": {
|
|
42
42
|
"@types/node": "22.19.19",
|
package/src/index.ts
CHANGED
|
@@ -22,6 +22,11 @@ export type { OpenAILanguageModelCompletionOptions } from './completion/openai-c
|
|
|
22
22
|
export type { OpenAIEmbeddingModelOptions } from './embedding/openai-embedding-model-options';
|
|
23
23
|
export type { OpenAISpeechModelOptions } from './speech/openai-speech-model-options';
|
|
24
24
|
export type { OpenAITranscriptionModelOptions } from './transcription/openai-transcription-model-options';
|
|
25
|
+
export { OpenAITranslationModel as Experimental_OpenAITranslationModel } from './translation/openai-translation-model';
|
|
26
|
+
export type {
|
|
27
|
+
OpenAITranslationModelId as Experimental_OpenAITranslationModelId,
|
|
28
|
+
OpenAITranslationModelOptions as Experimental_OpenAITranslationModelOptions,
|
|
29
|
+
} from './translation/openai-translation-model-options';
|
|
25
30
|
export type { OpenAIFilesOptions } from './files/openai-files-options';
|
|
26
31
|
export type {
|
|
27
32
|
OpenAIComputerAction,
|
package/src/openai-provider.ts
CHANGED
|
@@ -9,6 +9,7 @@ import type {
|
|
|
9
9
|
SpeechModelV4,
|
|
10
10
|
SkillsV4,
|
|
11
11
|
TranscriptionModelV4,
|
|
12
|
+
Experimental_SpeechTranslationModelV4 as SpeechTranslationModelV4,
|
|
12
13
|
} from '@ai-sdk/provider';
|
|
13
14
|
import {
|
|
14
15
|
loadApiKey,
|
|
@@ -36,6 +37,8 @@ import { OpenAISpeechModel } from './speech/openai-speech-model';
|
|
|
36
37
|
import type { OpenAISpeechModelId } from './speech/openai-speech-model-options';
|
|
37
38
|
import { OpenAITranscriptionModel } from './transcription/openai-transcription-model';
|
|
38
39
|
import type { OpenAITranscriptionModelId } from './transcription/openai-transcription-model-options';
|
|
40
|
+
import { OpenAITranslationModel } from './translation/openai-translation-model';
|
|
41
|
+
import type { OpenAITranslationModelId } from './translation/openai-translation-model-options';
|
|
39
42
|
import { OpenAISkills } from './skills/openai-skills';
|
|
40
43
|
import { VERSION } from './version';
|
|
41
44
|
|
|
@@ -97,6 +100,18 @@ export interface OpenAIProvider extends ProviderV4 {
|
|
|
97
100
|
*/
|
|
98
101
|
transcription(modelId: OpenAITranscriptionModelId): TranscriptionModelV4;
|
|
99
102
|
|
|
103
|
+
/**
|
|
104
|
+
* Creates an experimental model for streaming speech translation.
|
|
105
|
+
*/
|
|
106
|
+
translation(modelId: OpenAITranslationModelId): SpeechTranslationModelV4;
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Creates an experimental model for streaming speech translation.
|
|
110
|
+
*/
|
|
111
|
+
speechTranslationModel(
|
|
112
|
+
modelId: OpenAITranslationModelId,
|
|
113
|
+
): SpeechTranslationModelV4;
|
|
114
|
+
|
|
100
115
|
/**
|
|
101
116
|
* Creates a model for speech generation.
|
|
102
117
|
*/
|
|
@@ -242,6 +257,15 @@ export function createOpenAI(
|
|
|
242
257
|
webSocket: options.webSocket,
|
|
243
258
|
});
|
|
244
259
|
|
|
260
|
+
const createTranslationModel = (modelId: OpenAITranslationModelId) =>
|
|
261
|
+
new OpenAITranslationModel(modelId, {
|
|
262
|
+
provider: `${providerName}.translation`,
|
|
263
|
+
url: ({ path }) => `${baseURL}${path}`,
|
|
264
|
+
headers: getHeaders,
|
|
265
|
+
fetch: options.fetch,
|
|
266
|
+
webSocket: options.webSocket,
|
|
267
|
+
});
|
|
268
|
+
|
|
245
269
|
const createSpeechModel = (modelId: OpenAISpeechModelId) =>
|
|
246
270
|
new OpenAISpeechModel(modelId, {
|
|
247
271
|
provider: `${providerName}.speech`,
|
|
@@ -334,6 +358,9 @@ export function createOpenAI(
|
|
|
334
358
|
provider.transcription = createTranscriptionModel;
|
|
335
359
|
provider.transcriptionModel = createTranscriptionModel;
|
|
336
360
|
|
|
361
|
+
provider.translation = createTranslationModel;
|
|
362
|
+
provider.speechTranslationModel = createTranslationModel;
|
|
363
|
+
|
|
337
364
|
provider.speech = createSpeechModel;
|
|
338
365
|
provider.speechModel = createSpeechModel;
|
|
339
366
|
provider.files = createFiles;
|
|
@@ -497,7 +497,12 @@ export type OpenAIResponsesTool =
|
|
|
497
497
|
| {
|
|
498
498
|
type: 'web_search';
|
|
499
499
|
external_web_access: boolean | undefined;
|
|
500
|
-
filters:
|
|
500
|
+
filters:
|
|
501
|
+
| {
|
|
502
|
+
allowed_domains: string[] | undefined;
|
|
503
|
+
blocked_domains: string[] | undefined;
|
|
504
|
+
}
|
|
505
|
+
| undefined;
|
|
501
506
|
search_context_size: 'low' | 'medium' | 'high' | undefined;
|
|
502
507
|
user_location:
|
|
503
508
|
| {
|
|
@@ -204,7 +204,10 @@ export async function prepareResponsesTools({
|
|
|
204
204
|
type: 'web_search',
|
|
205
205
|
filters:
|
|
206
206
|
args.filters != null
|
|
207
|
-
? {
|
|
207
|
+
? {
|
|
208
|
+
allowed_domains: args.filters.allowedDomains,
|
|
209
|
+
blocked_domains: args.filters.blockedDomains,
|
|
210
|
+
}
|
|
208
211
|
: undefined,
|
|
209
212
|
external_web_access: args.externalWebAccess,
|
|
210
213
|
search_context_size: args.searchContextSize,
|
package/src/tool/web-search.ts
CHANGED
|
@@ -10,7 +10,10 @@ export const webSearchArgsSchema = lazySchema(() =>
|
|
|
10
10
|
z.object({
|
|
11
11
|
externalWebAccess: z.boolean().optional(),
|
|
12
12
|
filters: z
|
|
13
|
-
.object({
|
|
13
|
+
.object({
|
|
14
|
+
allowedDomains: z.array(z.string()).optional(),
|
|
15
|
+
blockedDomains: z.array(z.string()).optional(),
|
|
16
|
+
})
|
|
14
17
|
.optional(),
|
|
15
18
|
searchContextSize: z.enum(['low', 'medium', 'high']).optional(),
|
|
16
19
|
userLocation: z
|
|
@@ -140,8 +143,16 @@ export const webSearchToolFactory = createProviderExecutedToolFactory<
|
|
|
140
143
|
* Allowed domains for the search.
|
|
141
144
|
* If not provided, all domains are allowed.
|
|
142
145
|
* Subdomains of the provided domains are allowed as well.
|
|
146
|
+
* Omit the HTTP or HTTPS prefix. Maximum 100 domains.
|
|
143
147
|
*/
|
|
144
148
|
allowedDomains?: string[];
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Blocked domains for the search.
|
|
152
|
+
* Subdomains of the provided domains are blocked as well.
|
|
153
|
+
* Omit the HTTP or HTTPS prefix. Maximum 100 domains.
|
|
154
|
+
*/
|
|
155
|
+
blockedDomains?: string[];
|
|
145
156
|
};
|
|
146
157
|
|
|
147
158
|
/**
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import {
|
|
2
|
+
lazySchema,
|
|
3
|
+
zodSchema,
|
|
4
|
+
type InferSchema,
|
|
5
|
+
} from '@ai-sdk/provider-utils';
|
|
6
|
+
import { z } from 'zod/v4';
|
|
7
|
+
|
|
8
|
+
export type OpenAITranslationModelId = 'gpt-realtime-translate' | (string & {});
|
|
9
|
+
|
|
10
|
+
export const openAITranslationModelOptions = lazySchema(() =>
|
|
11
|
+
zodSchema(z.object({})),
|
|
12
|
+
);
|
|
13
|
+
|
|
14
|
+
export type OpenAITranslationModelOptions = InferSchema<
|
|
15
|
+
typeof openAITranslationModelOptions
|
|
16
|
+
>;
|
|
@@ -0,0 +1,397 @@
|
|
|
1
|
+
import {
|
|
2
|
+
InvalidArgumentError,
|
|
3
|
+
type Experimental_SpeechTranslationModelV4 as TranslationModelV4,
|
|
4
|
+
type Experimental_SpeechTranslationModelV4StreamOptions as SpeechTranslationModelV4StreamOptions,
|
|
5
|
+
type Experimental_SpeechTranslationModelV4StreamPart as SpeechTranslationModelV4StreamPart,
|
|
6
|
+
type SharedV4Warning,
|
|
7
|
+
} from '@ai-sdk/provider';
|
|
8
|
+
import {
|
|
9
|
+
combineHeaders,
|
|
10
|
+
connectToWebSocket,
|
|
11
|
+
convertToBase64,
|
|
12
|
+
parseProviderOptions,
|
|
13
|
+
safeParseJSON,
|
|
14
|
+
serializeModelOptions,
|
|
15
|
+
toWebSocketUrl,
|
|
16
|
+
WORKFLOW_DESERIALIZE,
|
|
17
|
+
WORKFLOW_SERIALIZE,
|
|
18
|
+
waitForWebSocketBufferDrain,
|
|
19
|
+
type WebSocketConnection,
|
|
20
|
+
type WebSocketLike,
|
|
21
|
+
} from '@ai-sdk/provider-utils';
|
|
22
|
+
import type { OpenAIConfig } from '../openai-config';
|
|
23
|
+
import {
|
|
24
|
+
openAITranslationModelOptions,
|
|
25
|
+
type OpenAITranslationModelId,
|
|
26
|
+
} from './openai-translation-model-options';
|
|
27
|
+
|
|
28
|
+
type OpenAIRealtimeTranslationEvent = {
|
|
29
|
+
type?: string;
|
|
30
|
+
delta?: string;
|
|
31
|
+
error?: { message?: string };
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
interface OpenAITranslationModelConfig extends OpenAIConfig {
|
|
35
|
+
_internal?: {
|
|
36
|
+
currentDate?: () => Date;
|
|
37
|
+
};
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export class OpenAITranslationModel implements TranslationModelV4 {
|
|
41
|
+
readonly specificationVersion = 'v4';
|
|
42
|
+
|
|
43
|
+
static [WORKFLOW_SERIALIZE](model: OpenAITranslationModel) {
|
|
44
|
+
return serializeModelOptions({
|
|
45
|
+
modelId: model.modelId,
|
|
46
|
+
config: model.config,
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
static [WORKFLOW_DESERIALIZE](options: {
|
|
51
|
+
modelId: OpenAITranslationModelId;
|
|
52
|
+
config: OpenAITranslationModelConfig;
|
|
53
|
+
}) {
|
|
54
|
+
return new OpenAITranslationModel(options.modelId, options.config);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
get provider(): string {
|
|
58
|
+
return this.config.provider;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
constructor(
|
|
62
|
+
readonly modelId: OpenAITranslationModelId,
|
|
63
|
+
private readonly config: OpenAITranslationModelConfig,
|
|
64
|
+
) {}
|
|
65
|
+
|
|
66
|
+
async doStream(
|
|
67
|
+
options: SpeechTranslationModelV4StreamOptions,
|
|
68
|
+
): Promise<Awaited<ReturnType<TranslationModelV4['doStream']>>> {
|
|
69
|
+
if (options.targetLanguage == null) {
|
|
70
|
+
throw new InvalidArgumentError({
|
|
71
|
+
argument: 'targetLanguage',
|
|
72
|
+
message: `targetLanguage is required for translation model '${this.modelId}'.`,
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const currentDate = this.config._internal?.currentDate?.() ?? new Date();
|
|
77
|
+
await parseProviderOptions({
|
|
78
|
+
provider: 'openai',
|
|
79
|
+
providerOptions: options.providerOptions,
|
|
80
|
+
schema: openAITranslationModelOptions,
|
|
81
|
+
});
|
|
82
|
+
const warnings: SharedV4Warning[] = [];
|
|
83
|
+
|
|
84
|
+
validateOpenAITranslationInputAudioFormat(options.inputAudioFormat);
|
|
85
|
+
|
|
86
|
+
if (options.sourceLanguage != null) {
|
|
87
|
+
warnings.push({
|
|
88
|
+
type: 'unsupported',
|
|
89
|
+
feature: 'sourceLanguage',
|
|
90
|
+
details:
|
|
91
|
+
'The OpenAI Realtime translation API auto-detects the source language and does not accept a source language.',
|
|
92
|
+
});
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
if (options.outputAudioFormat != null) {
|
|
96
|
+
warnings.push({
|
|
97
|
+
type: 'unsupported',
|
|
98
|
+
feature: 'outputAudioFormat',
|
|
99
|
+
details:
|
|
100
|
+
'The OpenAI Realtime translation API always outputs 24kHz 16-bit PCM audio and does not accept an output audio format.',
|
|
101
|
+
});
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
const headers = combineHeaders(this.config.headers?.(), options.headers);
|
|
105
|
+
const sessionUpdate = buildOpenAIRealtimeTranslationSession({
|
|
106
|
+
targetLanguage: options.targetLanguage,
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
return {
|
|
110
|
+
request: { body: sessionUpdate },
|
|
111
|
+
response: {
|
|
112
|
+
timestamp: currentDate,
|
|
113
|
+
modelId: this.modelId,
|
|
114
|
+
},
|
|
115
|
+
stream: createOpenAIRealtimeTranslationStream({
|
|
116
|
+
webSocket: this.config.webSocket,
|
|
117
|
+
url: toWebSocketUrl(
|
|
118
|
+
this.config.url({
|
|
119
|
+
path: `/realtime/translations?model=${encodeURIComponent(this.modelId)}`,
|
|
120
|
+
modelId: this.modelId,
|
|
121
|
+
}),
|
|
122
|
+
),
|
|
123
|
+
headers,
|
|
124
|
+
sessionUpdate,
|
|
125
|
+
warnings,
|
|
126
|
+
audio: options.audio,
|
|
127
|
+
abortSignal: options.abortSignal,
|
|
128
|
+
includeRawChunks: options.includeRawChunks,
|
|
129
|
+
}),
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function createOpenAIRealtimeTranslationStream({
|
|
135
|
+
webSocket,
|
|
136
|
+
url,
|
|
137
|
+
headers,
|
|
138
|
+
sessionUpdate,
|
|
139
|
+
warnings,
|
|
140
|
+
audio,
|
|
141
|
+
abortSignal,
|
|
142
|
+
includeRawChunks,
|
|
143
|
+
}: {
|
|
144
|
+
webSocket: OpenAIConfig['webSocket'];
|
|
145
|
+
url: URL;
|
|
146
|
+
headers: Record<string, string | undefined>;
|
|
147
|
+
sessionUpdate: unknown;
|
|
148
|
+
warnings: SharedV4Warning[];
|
|
149
|
+
audio: ReadableStream<Uint8Array | string>;
|
|
150
|
+
abortSignal: AbortSignal | undefined;
|
|
151
|
+
includeRawChunks: boolean | undefined;
|
|
152
|
+
}) {
|
|
153
|
+
let finished = false;
|
|
154
|
+
let cleanup: (closeCode?: number) => void = () => {};
|
|
155
|
+
|
|
156
|
+
return new ReadableStream<SpeechTranslationModelV4StreamPart>({
|
|
157
|
+
start: controller => {
|
|
158
|
+
const realtimeConnection = getOpenAIRealtimeConnection(headers);
|
|
159
|
+
let audioReader:
|
|
160
|
+
| ReadableStreamDefaultReader<Uint8Array | string>
|
|
161
|
+
| undefined;
|
|
162
|
+
let connection: WebSocketConnection | undefined;
|
|
163
|
+
|
|
164
|
+
let sourceText = '';
|
|
165
|
+
let translationText = '';
|
|
166
|
+
|
|
167
|
+
cleanup = (closeCode?: number) => {
|
|
168
|
+
if (audioReader != null) {
|
|
169
|
+
void audioReader.cancel().catch(() => {});
|
|
170
|
+
} else {
|
|
171
|
+
// pre-open failure or abort: cancel the caller's audio stream so an
|
|
172
|
+
// upstream producer piping into it does not hang:
|
|
173
|
+
void audio.cancel().catch(() => {});
|
|
174
|
+
}
|
|
175
|
+
connection?.close(closeCode);
|
|
176
|
+
};
|
|
177
|
+
|
|
178
|
+
const finishWithError = (error: unknown) => {
|
|
179
|
+
if (finished) return;
|
|
180
|
+
finished = true;
|
|
181
|
+
cleanup();
|
|
182
|
+
controller.error(error);
|
|
183
|
+
};
|
|
184
|
+
|
|
185
|
+
const finish = () => {
|
|
186
|
+
if (finished) return;
|
|
187
|
+
finished = true;
|
|
188
|
+
if (sourceText !== '') {
|
|
189
|
+
controller.enqueue({
|
|
190
|
+
type: 'source-transcript-final',
|
|
191
|
+
text: sourceText,
|
|
192
|
+
});
|
|
193
|
+
}
|
|
194
|
+
if (translationText !== '') {
|
|
195
|
+
controller.enqueue({
|
|
196
|
+
type: 'output-text-final',
|
|
197
|
+
text: translationText,
|
|
198
|
+
});
|
|
199
|
+
}
|
|
200
|
+
controller.enqueue({
|
|
201
|
+
type: 'finish',
|
|
202
|
+
sourceText,
|
|
203
|
+
outputText: translationText,
|
|
204
|
+
usage: undefined,
|
|
205
|
+
});
|
|
206
|
+
controller.close();
|
|
207
|
+
cleanup(1000);
|
|
208
|
+
};
|
|
209
|
+
|
|
210
|
+
const sendAudio = async (socket: WebSocketLike) => {
|
|
211
|
+
audioReader = audio.getReader();
|
|
212
|
+
try {
|
|
213
|
+
while (true) {
|
|
214
|
+
const { done, value } = await audioReader.read();
|
|
215
|
+
if (done || finished) break;
|
|
216
|
+
socket.send(
|
|
217
|
+
JSON.stringify({
|
|
218
|
+
type: 'session.input_audio_buffer.append',
|
|
219
|
+
audio: convertToBase64(value),
|
|
220
|
+
}),
|
|
221
|
+
);
|
|
222
|
+
// backpressure: pause reads while the socket buffer is full
|
|
223
|
+
await waitForWebSocketBufferDrain(socket);
|
|
224
|
+
}
|
|
225
|
+
} finally {
|
|
226
|
+
audioReader.releaseLock();
|
|
227
|
+
// unlocked again: cleanup must cancel `audio`, not the reader
|
|
228
|
+
audioReader = undefined;
|
|
229
|
+
}
|
|
230
|
+
if (!finished) {
|
|
231
|
+
socket.send(JSON.stringify({ type: 'session.close' }));
|
|
232
|
+
}
|
|
233
|
+
};
|
|
234
|
+
|
|
235
|
+
connection = connectToWebSocket({
|
|
236
|
+
url,
|
|
237
|
+
protocols: realtimeConnection.protocols,
|
|
238
|
+
headers: realtimeConnection.headers,
|
|
239
|
+
webSocket,
|
|
240
|
+
abortSignal,
|
|
241
|
+
onAbort: finishWithError,
|
|
242
|
+
onProcessingError: finishWithError,
|
|
243
|
+
onOpen: socket => {
|
|
244
|
+
controller.enqueue({ type: 'stream-start', warnings });
|
|
245
|
+
socket.send(JSON.stringify(sessionUpdate));
|
|
246
|
+
void sendAudio(socket).catch(finishWithError);
|
|
247
|
+
},
|
|
248
|
+
onMessageText: async text => {
|
|
249
|
+
if (finished) return;
|
|
250
|
+
const parsed = await safeParseJSON({ text });
|
|
251
|
+
if (!parsed.success) return;
|
|
252
|
+
const raw = parsed.value as OpenAIRealtimeTranslationEvent;
|
|
253
|
+
|
|
254
|
+
if (includeRawChunks) {
|
|
255
|
+
controller.enqueue({ type: 'raw', rawValue: raw });
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
switch (raw.type) {
|
|
259
|
+
case 'session.output_audio.delta': {
|
|
260
|
+
// skip empty deltas: an empty `audio` part carries no data
|
|
261
|
+
if (raw.delta) {
|
|
262
|
+
controller.enqueue({
|
|
263
|
+
type: 'audio',
|
|
264
|
+
audio: raw.delta,
|
|
265
|
+
});
|
|
266
|
+
}
|
|
267
|
+
break;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
case 'session.output_transcript.delta': {
|
|
271
|
+
translationText += raw.delta ?? '';
|
|
272
|
+
controller.enqueue({
|
|
273
|
+
type: 'output-text-delta',
|
|
274
|
+
delta: raw.delta ?? '',
|
|
275
|
+
});
|
|
276
|
+
break;
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
case 'session.input_transcript.delta': {
|
|
280
|
+
sourceText += raw.delta ?? '';
|
|
281
|
+
controller.enqueue({
|
|
282
|
+
type: 'source-transcript-delta',
|
|
283
|
+
delta: raw.delta ?? '',
|
|
284
|
+
});
|
|
285
|
+
break;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
case 'session.closed': {
|
|
289
|
+
finish();
|
|
290
|
+
break;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
case 'error': {
|
|
294
|
+
controller.enqueue({
|
|
295
|
+
type: 'error',
|
|
296
|
+
error: new Error(raw.error?.message ?? 'OpenAI realtime error'),
|
|
297
|
+
});
|
|
298
|
+
break;
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
},
|
|
302
|
+
onSocketError: () => {
|
|
303
|
+
finishWithError(new Error('OpenAI realtime translation error'));
|
|
304
|
+
},
|
|
305
|
+
onClose: ({ code, reason }) => {
|
|
306
|
+
if (finished) return;
|
|
307
|
+
// a close before the finish event is an abnormal termination:
|
|
308
|
+
// surface the close diagnostics instead of silently closing
|
|
309
|
+
finishWithError(
|
|
310
|
+
new Error(
|
|
311
|
+
`OpenAI realtime translation WebSocket closed unexpectedly before finishing` +
|
|
312
|
+
` (code ${code ?? 'unknown'}${reason ? `, reason: ${reason}` : ''}).`,
|
|
313
|
+
),
|
|
314
|
+
);
|
|
315
|
+
},
|
|
316
|
+
});
|
|
317
|
+
},
|
|
318
|
+
|
|
319
|
+
cancel: () => {
|
|
320
|
+
if (finished) return;
|
|
321
|
+
finished = true;
|
|
322
|
+
cleanup();
|
|
323
|
+
},
|
|
324
|
+
});
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
function buildOpenAIRealtimeTranslationSession({
|
|
328
|
+
targetLanguage,
|
|
329
|
+
}: {
|
|
330
|
+
targetLanguage: string;
|
|
331
|
+
}) {
|
|
332
|
+
return {
|
|
333
|
+
type: 'session.update',
|
|
334
|
+
session: {
|
|
335
|
+
audio: {
|
|
336
|
+
input: {
|
|
337
|
+
transcription: {
|
|
338
|
+
model: 'gpt-realtime-whisper',
|
|
339
|
+
},
|
|
340
|
+
noise_reduction: null,
|
|
341
|
+
},
|
|
342
|
+
output: {
|
|
343
|
+
language: targetLanguage,
|
|
344
|
+
},
|
|
345
|
+
},
|
|
346
|
+
},
|
|
347
|
+
};
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
function validateOpenAITranslationInputAudioFormat(
|
|
351
|
+
inputAudioFormat: SpeechTranslationModelV4StreamOptions['inputAudioFormat'],
|
|
352
|
+
) {
|
|
353
|
+
if (
|
|
354
|
+
inputAudioFormat.type !== 'audio/pcm' ||
|
|
355
|
+
(inputAudioFormat.rate != null && inputAudioFormat.rate !== 24000)
|
|
356
|
+
) {
|
|
357
|
+
throw new InvalidArgumentError({
|
|
358
|
+
argument: 'inputAudioFormat',
|
|
359
|
+
message:
|
|
360
|
+
'The OpenAI Realtime translation API only supports 24kHz 16-bit PCM input audio.',
|
|
361
|
+
});
|
|
362
|
+
}
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
// The bearer token rides the `openai-insecure-api-key` subprotocol (native
|
|
366
|
+
// `WebSocket` cannot send headers) and the Authorization header is stripped:
|
|
367
|
+
// OpenAI rejects handshakes that send both auth channels.
|
|
368
|
+
function getOpenAIRealtimeConnection(
|
|
369
|
+
headers: Record<string, string | undefined>,
|
|
370
|
+
): {
|
|
371
|
+
protocols: string[];
|
|
372
|
+
headers: Record<string, string | undefined>;
|
|
373
|
+
} {
|
|
374
|
+
// last case-variant wins: combineHeaders keeps case-distinct keys and
|
|
375
|
+
// spreads per-call headers after configuration headers
|
|
376
|
+
let authorization: string | undefined;
|
|
377
|
+
for (const [key, value] of Object.entries(headers)) {
|
|
378
|
+
if (key.toLowerCase() === 'authorization' && value != null) {
|
|
379
|
+
authorization = value;
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
// the HTTP auth scheme is case-insensitive
|
|
383
|
+
const token = authorization?.match(/^bearer\s+(.+)$/i)?.[1];
|
|
384
|
+
|
|
385
|
+
if (token == null) {
|
|
386
|
+
return { protocols: ['realtime'], headers };
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
return {
|
|
390
|
+
protocols: ['realtime', `openai-insecure-api-key.${token}`],
|
|
391
|
+
headers: Object.fromEntries(
|
|
392
|
+
Object.entries(headers).filter(
|
|
393
|
+
([key]) => key.toLowerCase() !== 'authorization',
|
|
394
|
+
),
|
|
395
|
+
),
|
|
396
|
+
};
|
|
397
|
+
}
|