@ai-sdk/provider 4.0.13 → 4.0.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,26 @@
1
1
  # @ai-sdk/provider
2
2
 
3
+ ## 4.0.15
4
+
5
+ ### Patch Changes
6
+
7
+ - 5c0054d: Add optional browser-direct WebRTC for experimental client-delegated Live conversations alongside the existing WebSocket path. Exchange SDP through an application endpoint with `api.session`, configure server-owned data-channel permissions, and preserve committed React session ownership. Capture follows the selected sender track, borrowed tracks remain caller-owned, and disconnect recovery and finalization stay bounded. Applications continue to handle client delegation and submit context; Live session updates and Responses delegation remain unsupported.
8
+
9
+ Serialize microphone sender changes and close the peer if detachment fails, without stopping borrowed tracks. Validate nonempty SDP setup answers with a bounded response body, and document the same-origin broker authentication contract.
10
+
11
+ - 39535af: Add experimental OpenAI Live provider support through the unified `openai.experimental_realtime` factory for server WebSocket sessions with client delegation. Applications own their agents and tools, receive continuous audio/transcript events and delegation metadata, and return context through validated channels. Support immutable startup options, microphone mute controls, graceful session close, and cumulative voice usage. Responses delegation and Live session updates reject before sending.
12
+
13
+ Route known Live model IDs to Live, allow an OpenAI-specific `api` override for early-access models, and preserve legacy Realtime defaults for unknown IDs. Token minting follows the same selection rules and rejects Live before requesting unsupported credentials. Extend the realtime v4 specification with optional server WebSocket configuration, per-connection raw-event parsers, and model-wide startup/finalization capabilities. This provider layer supplies connection settings and protocol mapping for server adapters; browser lifecycle and UI integration belong to the core runtime and framework hooks.
14
+
15
+ Preserve Realtime client event IDs for session updates and audio appends, and correlate server errors with the originating client event.
16
+
17
+ ## 4.0.14
18
+
19
+ ### Patch Changes
20
+
21
+ - 5ec21a6: fix: reject unsupported batch request types
22
+ - 7469a3b: feat: support image generation requests in batches
23
+
3
24
  ## 4.0.13
4
25
 
5
26
  ### Patch Changes
package/dist/index.d.ts CHANGED
@@ -736,6 +736,247 @@ type LanguageModelV4GenerateResult = {
736
736
  warnings: Array<SharedV4Warning>;
737
737
  };
738
738
 
739
+ /**
740
+ * An image file that can be used for image editing or variation generation.
741
+ */
742
+ type ImageModelV4File = {
743
+ type: 'file';
744
+ /**
745
+ * The IANA media type of the file, e.g. `image/png`. Any string is supported.
746
+ *
747
+ * @see https://www.iana.org/assignments/media-types/media-types.xhtml
748
+ */
749
+ mediaType: string;
750
+ /**
751
+ * Generated file data as base64 encoded strings or binary data.
752
+ *
753
+ * The file data should be returned without any unnecessary conversion.
754
+ * If the API returns base64 encoded strings, the file data should be returned
755
+ * as base64 encoded strings. If the API returns binary data, the file data should
756
+ * be returned as binary data.
757
+ */
758
+ data: string | Uint8Array;
759
+ /**
760
+ * Optional provider-specific metadata for the file part.
761
+ */
762
+ providerOptions?: SharedV4ProviderMetadata;
763
+ } | {
764
+ type: 'url';
765
+ /**
766
+ * The URL of the image file.
767
+ */
768
+ url: string;
769
+ /**
770
+ * Optional provider-specific metadata for the file part.
771
+ */
772
+ providerOptions?: SharedV4ProviderMetadata;
773
+ };
774
+
775
+ type ImageModelV4CallOptions = {
776
+ /**
777
+ * Prompt for the image generation. Some operations, like upscaling, may not require a prompt.
778
+ */
779
+ prompt: string | undefined;
780
+ /**
781
+ * Number of images to generate.
782
+ */
783
+ n: number;
784
+ /**
785
+ * Size of the images to generate.
786
+ * Must have the format `{width}x{height}`.
787
+ * `undefined` will use the provider's default size.
788
+ */
789
+ size: `${number}x${number}` | undefined;
790
+ /**
791
+ * Aspect ratio of the images to generate.
792
+ * Must have the format `{width}:{height}`.
793
+ * `undefined` will use the provider's default aspect ratio.
794
+ */
795
+ aspectRatio: `${number}:${number}` | undefined;
796
+ /**
797
+ * Seed for the image generation.
798
+ * `undefined` will use the provider's default seed.
799
+ */
800
+ seed: number | undefined;
801
+ /**
802
+ * Array of images for image editing or variation generation.
803
+ * The images should be provided as base64 encoded strings or binary data.
804
+ */
805
+ files: ImageModelV4File[] | undefined;
806
+ /**
807
+ * Mask image for inpainting operations.
808
+ * The mask should be provided as base64 encoded strings or binary data.
809
+ */
810
+ mask: ImageModelV4File | undefined;
811
+ /**
812
+ * Additional provider-specific options that are passed through to the provider
813
+ * as body parameters.
814
+ *
815
+ * The outer record is keyed by the provider name, and the inner
816
+ * record is keyed by the provider-specific metadata key.
817
+ *
818
+ * ```ts
819
+ * {
820
+ * "openai": {
821
+ * "style": "vivid"
822
+ * }
823
+ * }
824
+ * ```
825
+ */
826
+ providerOptions: SharedV4ProviderOptions;
827
+ /**
828
+ * Abort signal for cancelling the operation.
829
+ */
830
+ abortSignal?: AbortSignal;
831
+ /**
832
+ * Additional HTTP headers to be sent with the request.
833
+ * Only applicable for HTTP-based providers.
834
+ */
835
+ headers?: Record<string, string | undefined>;
836
+ };
837
+
838
+ /**
839
+ * Usage information for an image model call.
840
+ */
841
+ type ImageModelV4Usage = {
842
+ /**
843
+ * The number of input (prompt) tokens used.
844
+ */
845
+ inputTokens: number | undefined;
846
+ /**
847
+ * The number of output tokens used, if reported by the provider.
848
+ */
849
+ outputTokens: number | undefined;
850
+ /**
851
+ * The total number of tokens as reported by the provider.
852
+ */
853
+ totalTokens: number | undefined;
854
+ };
855
+
856
+ type ImageModelV4ProviderMetadata = Record<string, {
857
+ images: JSONArray;
858
+ } & JSONValue>;
859
+ /**
860
+ * The result of an image model doGenerate call.
861
+ */
862
+ type ImageModelV4Result = {
863
+ /**
864
+ * Generated images as base64 encoded strings or binary data.
865
+ * The images should be returned without any unnecessary conversion.
866
+ * If the API returns base64 encoded strings, the images should be returned
867
+ * as base64 encoded strings. If the API returns binary data, the images should
868
+ * be returned as binary data.
869
+ */
870
+ images: Array<string> | Array<Uint8Array>;
871
+ /**
872
+ * Whether an unsuccessful result, such as an empty image result, can be
873
+ * retried. When omitted, the result is unclassified.
874
+ */
875
+ isRetryable?: boolean;
876
+ /**
877
+ * Warnings for the call, e.g. unsupported features.
878
+ */
879
+ warnings: Array<SharedV4Warning>;
880
+ /**
881
+ * Additional provider-specific metadata. They are passed through
882
+ * from the provider to the AI SDK and enable provider-specific
883
+ * results that can be fully encapsulated in the provider.
884
+ *
885
+ * The outer record is keyed by the provider name, and the inner
886
+ * record is provider-specific metadata. It always includes an
887
+ * `images` key with image-specific metadata
888
+ *
889
+ * ```ts
890
+ * {
891
+ * "openai": {
892
+ * "images": ["revisedPrompt": "Revised prompt here."]
893
+ * }
894
+ * }
895
+ * ```
896
+ */
897
+ providerMetadata?: ImageModelV4ProviderMetadata;
898
+ /**
899
+ * Response information for telemetry and debugging purposes.
900
+ */
901
+ response: {
902
+ /**
903
+ * Timestamp for the start of the generated response.
904
+ */
905
+ timestamp: Date;
906
+ /**
907
+ * The ID of the response model that was used to generate the response.
908
+ */
909
+ modelId: string;
910
+ /**
911
+ * Response headers.
912
+ */
913
+ headers: Record<string, string> | undefined;
914
+ };
915
+ /**
916
+ * Optional token usage for the image generation call (if the provider reports it).
917
+ */
918
+ usage?: ImageModelV4Usage;
919
+ };
920
+
921
+ type GetMaxImagesPerCallFunction$2 = (options: {
922
+ modelId: string;
923
+ }) => PromiseLike<number | undefined> | number | undefined;
924
+ /**
925
+ * Image generation model specification version 4.
926
+ */
927
+ type ImageModelV4 = {
928
+ /**
929
+ * The image model must specify which image model interface
930
+ * version it implements. This will allow us to evolve the image
931
+ * model interface and retain backwards compatibility. The different
932
+ * implementation versions can be handled as a discriminated union
933
+ * on our side.
934
+ */
935
+ readonly specificationVersion: 'v4';
936
+ /**
937
+ * Name of the provider for logging purposes.
938
+ */
939
+ readonly provider: string;
940
+ /**
941
+ * Provider-specific model ID for logging purposes.
942
+ */
943
+ readonly modelId: string;
944
+ /**
945
+ * Limit of how many images can be generated in a single API call.
946
+ * Can be set to a number for a fixed limit, to undefined to use
947
+ * the global limit, or a function that returns a number or undefined,
948
+ * optionally as a promise.
949
+ */
950
+ readonly maxImagesPerCall: number | undefined | GetMaxImagesPerCallFunction$2;
951
+ /**
952
+ * Generates an array of images.
953
+ */
954
+ doGenerate(options: ImageModelV4CallOptions): PromiseLike<ImageModelV4Result>;
955
+ };
956
+
957
+ /**
958
+ * Fields shared by every request in a batch.
959
+ */
960
+ type BatchV4RequestBase<ModelId extends string = string> = {
961
+ /**
962
+ * Application-provided identifier used to correlate the request with its
963
+ * result.
964
+ */
965
+ readonly id: string;
966
+ /**
967
+ * Provider-specific model ID for this request.
968
+ */
969
+ readonly modelId: ModelId;
970
+ };
971
+
972
+ /**
973
+ * One image generation request in a batch.
974
+ */
975
+ type ImageBatchV4Request<ModelId extends string = string> = BatchV4RequestBase<ModelId> & {
976
+ readonly type: 'image';
977
+ readonly options: Pick<ImageModelV4CallOptions, 'prompt' | 'n' | 'size' | 'aspectRatio' | 'seed' | 'files' | 'mask' | 'providerOptions'>;
978
+ };
979
+
739
980
  /**
740
981
  * A tool has a name, a description, and a set of parameters.
741
982
  *
@@ -1249,21 +1490,6 @@ type LanguageModelV4CallOptions = {
1249
1490
  providerOptions?: SharedV4ProviderOptions;
1250
1491
  };
1251
1492
 
1252
- /**
1253
- * Fields shared by every request in a batch.
1254
- */
1255
- type BatchV4RequestBase<ModelId extends string = string> = {
1256
- /**
1257
- * Application-provided identifier used to correlate the request with its
1258
- * result.
1259
- */
1260
- readonly id: string;
1261
- /**
1262
- * Provider-specific model ID for this request.
1263
- */
1264
- readonly modelId: ModelId;
1265
- };
1266
-
1267
1493
  /**
1268
1494
  * A normalized text generation request within a batch.
1269
1495
  */
@@ -1278,7 +1504,8 @@ type TextBatchV4Request<ModelId extends string = string> = BatchV4RequestBase<Mo
1278
1504
  * Additional modality-specific model ID types can be added to this mapping.
1279
1505
  */
1280
1506
  type BatchV4ModelIds = {
1281
- readonly text: string;
1507
+ readonly text?: string;
1508
+ readonly image?: string;
1282
1509
  };
1283
1510
  type BatchV4CallOptions = {
1284
1511
  readonly providerOptions?: SharedV4ProviderOptions;
@@ -1319,7 +1546,7 @@ type BatchV4Status = {
1319
1546
  readonly expiresAt?: string;
1320
1547
  readonly providerMetadata?: SharedV4ProviderMetadata;
1321
1548
  };
1322
- type BatchV4Request<ModelIds extends BatchV4ModelIds = BatchV4ModelIds> = TextBatchV4Request<ModelIds['text']>;
1549
+ type BatchV4Request<ModelIds extends BatchV4ModelIds = BatchV4ModelIds> = TextBatchV4Request<ModelIds['text'] & string> | ImageBatchV4Request<ModelIds['image'] & string>;
1323
1550
  /**
1324
1551
  * Options for starting a batch of requests discriminated by modality.
1325
1552
  *
@@ -1393,13 +1620,19 @@ type BatchV4ItemResultBase<RESULT> = {
1393
1620
  type TextBatchV4ItemResult = {
1394
1621
  readonly type: 'text';
1395
1622
  } & BatchV4ItemResultBase<LanguageModelV4GenerateResult>;
1623
+ /**
1624
+ * A complete terminal result for one request in an image batch.
1625
+ */
1626
+ type ImageBatchV4ItemResult = {
1627
+ readonly type: 'image';
1628
+ } & BatchV4ItemResultBase<ImageModelV4Result>;
1396
1629
  /**
1397
1630
  * A complete terminal result for one request in a batch, discriminated by
1398
1631
  * modality.
1399
1632
  *
1400
1633
  * Additional modality-specific item results can be added to this union.
1401
1634
  */
1402
- type BatchV4ItemResult = TextBatchV4ItemResult;
1635
+ type BatchV4ItemResult = TextBatchV4ItemResult | ImageBatchV4ItemResult;
1403
1636
  /**
1404
1637
  * Specification for a batch interface that implements batch interface version 4.
1405
1638
  */
@@ -2294,224 +2527,6 @@ type FilesV4 = {
2294
2527
  deleteFile?(options: FilesV4DeleteFileCallOptions): PromiseLike<FilesV4DeleteFileResult>;
2295
2528
  };
2296
2529
 
2297
- /**
2298
- * An image file that can be used for image editing or variation generation.
2299
- */
2300
- type ImageModelV4File = {
2301
- type: 'file';
2302
- /**
2303
- * The IANA media type of the file, e.g. `image/png`. Any string is supported.
2304
- *
2305
- * @see https://www.iana.org/assignments/media-types/media-types.xhtml
2306
- */
2307
- mediaType: string;
2308
- /**
2309
- * Generated file data as base64 encoded strings or binary data.
2310
- *
2311
- * The file data should be returned without any unnecessary conversion.
2312
- * If the API returns base64 encoded strings, the file data should be returned
2313
- * as base64 encoded strings. If the API returns binary data, the file data should
2314
- * be returned as binary data.
2315
- */
2316
- data: string | Uint8Array;
2317
- /**
2318
- * Optional provider-specific metadata for the file part.
2319
- */
2320
- providerOptions?: SharedV4ProviderMetadata;
2321
- } | {
2322
- type: 'url';
2323
- /**
2324
- * The URL of the image file.
2325
- */
2326
- url: string;
2327
- /**
2328
- * Optional provider-specific metadata for the file part.
2329
- */
2330
- providerOptions?: SharedV4ProviderMetadata;
2331
- };
2332
-
2333
- type ImageModelV4CallOptions = {
2334
- /**
2335
- * Prompt for the image generation. Some operations, like upscaling, may not require a prompt.
2336
- */
2337
- prompt: string | undefined;
2338
- /**
2339
- * Number of images to generate.
2340
- */
2341
- n: number;
2342
- /**
2343
- * Size of the images to generate.
2344
- * Must have the format `{width}x{height}`.
2345
- * `undefined` will use the provider's default size.
2346
- */
2347
- size: `${number}x${number}` | undefined;
2348
- /**
2349
- * Aspect ratio of the images to generate.
2350
- * Must have the format `{width}:{height}`.
2351
- * `undefined` will use the provider's default aspect ratio.
2352
- */
2353
- aspectRatio: `${number}:${number}` | undefined;
2354
- /**
2355
- * Seed for the image generation.
2356
- * `undefined` will use the provider's default seed.
2357
- */
2358
- seed: number | undefined;
2359
- /**
2360
- * Array of images for image editing or variation generation.
2361
- * The images should be provided as base64 encoded strings or binary data.
2362
- */
2363
- files: ImageModelV4File[] | undefined;
2364
- /**
2365
- * Mask image for inpainting operations.
2366
- * The mask should be provided as base64 encoded strings or binary data.
2367
- */
2368
- mask: ImageModelV4File | undefined;
2369
- /**
2370
- * Additional provider-specific options that are passed through to the provider
2371
- * as body parameters.
2372
- *
2373
- * The outer record is keyed by the provider name, and the inner
2374
- * record is keyed by the provider-specific metadata key.
2375
- *
2376
- * ```ts
2377
- * {
2378
- * "openai": {
2379
- * "style": "vivid"
2380
- * }
2381
- * }
2382
- * ```
2383
- */
2384
- providerOptions: SharedV4ProviderOptions;
2385
- /**
2386
- * Abort signal for cancelling the operation.
2387
- */
2388
- abortSignal?: AbortSignal;
2389
- /**
2390
- * Additional HTTP headers to be sent with the request.
2391
- * Only applicable for HTTP-based providers.
2392
- */
2393
- headers?: Record<string, string | undefined>;
2394
- };
2395
-
2396
- /**
2397
- * Usage information for an image model call.
2398
- */
2399
- type ImageModelV4Usage = {
2400
- /**
2401
- * The number of input (prompt) tokens used.
2402
- */
2403
- inputTokens: number | undefined;
2404
- /**
2405
- * The number of output tokens used, if reported by the provider.
2406
- */
2407
- outputTokens: number | undefined;
2408
- /**
2409
- * The total number of tokens as reported by the provider.
2410
- */
2411
- totalTokens: number | undefined;
2412
- };
2413
-
2414
- type ImageModelV4ProviderMetadata = Record<string, {
2415
- images: JSONArray;
2416
- } & JSONValue>;
2417
- /**
2418
- * The result of an image model doGenerate call.
2419
- */
2420
- type ImageModelV4Result = {
2421
- /**
2422
- * Generated images as base64 encoded strings or binary data.
2423
- * The images should be returned without any unnecessary conversion.
2424
- * If the API returns base64 encoded strings, the images should be returned
2425
- * as base64 encoded strings. If the API returns binary data, the images should
2426
- * be returned as binary data.
2427
- */
2428
- images: Array<string> | Array<Uint8Array>;
2429
- /**
2430
- * Whether an unsuccessful result, such as an empty image result, can be
2431
- * retried. When omitted, the result is unclassified.
2432
- */
2433
- isRetryable?: boolean;
2434
- /**
2435
- * Warnings for the call, e.g. unsupported features.
2436
- */
2437
- warnings: Array<SharedV4Warning>;
2438
- /**
2439
- * Additional provider-specific metadata. They are passed through
2440
- * from the provider to the AI SDK and enable provider-specific
2441
- * results that can be fully encapsulated in the provider.
2442
- *
2443
- * The outer record is keyed by the provider name, and the inner
2444
- * record is provider-specific metadata. It always includes an
2445
- * `images` key with image-specific metadata
2446
- *
2447
- * ```ts
2448
- * {
2449
- * "openai": {
2450
- * "images": ["revisedPrompt": "Revised prompt here."]
2451
- * }
2452
- * }
2453
- * ```
2454
- */
2455
- providerMetadata?: ImageModelV4ProviderMetadata;
2456
- /**
2457
- * Response information for telemetry and debugging purposes.
2458
- */
2459
- response: {
2460
- /**
2461
- * Timestamp for the start of the generated response.
2462
- */
2463
- timestamp: Date;
2464
- /**
2465
- * The ID of the response model that was used to generate the response.
2466
- */
2467
- modelId: string;
2468
- /**
2469
- * Response headers.
2470
- */
2471
- headers: Record<string, string> | undefined;
2472
- };
2473
- /**
2474
- * Optional token usage for the image generation call (if the provider reports it).
2475
- */
2476
- usage?: ImageModelV4Usage;
2477
- };
2478
-
2479
- type GetMaxImagesPerCallFunction$2 = (options: {
2480
- modelId: string;
2481
- }) => PromiseLike<number | undefined> | number | undefined;
2482
- /**
2483
- * Image generation model specification version 4.
2484
- */
2485
- type ImageModelV4 = {
2486
- /**
2487
- * The image model must specify which image model interface
2488
- * version it implements. This will allow us to evolve the image
2489
- * model interface and retain backwards compatibility. The different
2490
- * implementation versions can be handled as a discriminated union
2491
- * on our side.
2492
- */
2493
- readonly specificationVersion: 'v4';
2494
- /**
2495
- * Name of the provider for logging purposes.
2496
- */
2497
- readonly provider: string;
2498
- /**
2499
- * Provider-specific model ID for logging purposes.
2500
- */
2501
- readonly modelId: string;
2502
- /**
2503
- * Limit of how many images can be generated in a single API call.
2504
- * Can be set to a number for a fixed limit, to undefined to use
2505
- * the global limit, or a function that returns a number or undefined,
2506
- * optionally as a promise.
2507
- */
2508
- readonly maxImagesPerCall: number | undefined | GetMaxImagesPerCallFunction$2;
2509
- /**
2510
- * Generates an array of images.
2511
- */
2512
- doGenerate(options: ImageModelV4CallOptions): PromiseLike<ImageModelV4Result>;
2513
- };
2514
-
2515
2530
  /**
2516
2531
  * Usage information for an image model call.
2517
2532
  */
@@ -6982,12 +6997,33 @@ type RealtimeModelV4FunctionCallOutput = {
6982
6997
  type RealtimeModelV4ClientEvent = {
6983
6998
  type: 'session-update';
6984
6999
  config: RealtimeModelV4SessionConfig;
7000
+ eventId?: string;
7001
+ } | {
7002
+ type: 'session-start';
7003
+ config: RealtimeModelV4SessionConfig;
7004
+ eventId?: string;
7005
+ } | {
7006
+ type: 'session-close';
7007
+ eventId?: string;
7008
+ } | {
7009
+ type: 'input-audio-mute';
7010
+ eventId?: string;
7011
+ } | {
7012
+ type: 'input-audio-unmute';
7013
+ eventId?: string;
7014
+ } | {
7015
+ type: 'context-append';
7016
+ content: string;
7017
+ delegationId: string | null;
7018
+ eventId?: string;
7019
+ providerOptions?: SharedV4ProviderOptions;
6985
7020
  } | {
6986
7021
  type: 'input-audio-append';
6987
7022
  /**
6988
7023
  * Base64-encoded audio chunk to append to the input buffer.
6989
7024
  */
6990
7025
  audio: string;
7026
+ eventId?: string;
6991
7027
  } | {
6992
7028
  type: 'input-audio-commit';
6993
7029
  } | {
@@ -7028,6 +7064,51 @@ type RealtimeModelV4ClientEvent = {
7028
7064
  * event data for debugging and provider-specific access.
7029
7065
  */
7030
7066
  type RealtimeModelV4ServerEvent = {
7067
+ type: 'session-started';
7068
+ sessionId: string;
7069
+ delegationMode?: 'client' | 'provider';
7070
+ raw: unknown;
7071
+ } | {
7072
+ type: 'session-closed';
7073
+ sessionId?: string;
7074
+ usage: {
7075
+ seconds: number;
7076
+ };
7077
+ reason: string;
7078
+ raw: unknown;
7079
+ } | {
7080
+ type: 'session-usage';
7081
+ /** Cumulative duration snapshot, not an increment. */
7082
+ usage: {
7083
+ seconds: number;
7084
+ };
7085
+ contextWindowUsageRatio?: number;
7086
+ raw: unknown;
7087
+ } | {
7088
+ type: 'audio-chunk';
7089
+ delta: string;
7090
+ raw: unknown;
7091
+ } | {
7092
+ type: 'transcript-fragment';
7093
+ speaker: 'user' | 'assistant';
7094
+ delta: string;
7095
+ startMs: number;
7096
+ endMs: number;
7097
+ raw: unknown;
7098
+ } | {
7099
+ type: 'delegation-created';
7100
+ delegationId: string;
7101
+ target?: 'client' | 'provider';
7102
+ offsetMs?: number;
7103
+ responseId?: string;
7104
+ raw: unknown;
7105
+ } | {
7106
+ type: 'command-acknowledged';
7107
+ /** Provider-native command name that was acknowledged. */
7108
+ command: string;
7109
+ clientEventId?: string;
7110
+ raw: unknown;
7111
+ } | {
7031
7112
  type: 'session-created';
7032
7113
  sessionId?: string;
7033
7114
  raw: unknown;
@@ -7158,6 +7239,7 @@ type RealtimeModelV4ServerEvent = {
7158
7239
  type: 'error';
7159
7240
  message: string;
7160
7241
  code?: string;
7242
+ clientEventId?: string;
7161
7243
  raw: unknown;
7162
7244
  } | {
7163
7245
  type: 'custom';
@@ -7188,6 +7270,38 @@ type RealtimeModelV4 = {
7188
7270
  * Provider-specific model ID (e.g. 'gpt-4o-realtime', 'grok-3').
7189
7271
  */
7190
7272
  readonly modelId: string;
7273
+ /** Conversation semantics and supported transports, when declared. */
7274
+ readonly capabilities?: {
7275
+ conversation: 'continuous' | 'turn-based';
7276
+ transports: readonly ('websocket' | 'webrtc')[];
7277
+ /** Omission preserves the legacy client-secret WebSocket connection. */
7278
+ connections?: readonly ('client-secret-websocket' | 'server-websocket' | 'webrtc')[];
7279
+ /** Omission preserves session-update startup. WebRTC setup may start the session. */
7280
+ startup?: 'session-start' | 'session-update';
7281
+ /** Omission preserves transport-close finalization. */
7282
+ finalization?: 'session-close' | 'transport-close';
7283
+ };
7284
+ /** Server-only connection settings. Headers can contain long-lived credentials. */
7285
+ getServerWebSocketConfig?(): {
7286
+ url: string;
7287
+ headers: Record<string, string>;
7288
+ } | PromiseLike<{
7289
+ url: string;
7290
+ headers: Record<string, string>;
7291
+ }>;
7292
+ /** Provider-specific setup for the WebRTC event channel. */
7293
+ getWebRTCConfig?(): {
7294
+ dataChannelLabel: string;
7295
+ };
7296
+ /** Server-side SDP exchange. Return only the answer and session ID to the client. */
7297
+ doCreateWebRTCSession?(options: {
7298
+ sdp: string;
7299
+ sessionConfig?: RealtimeModelV4SessionConfig;
7300
+ abortSignal?: AbortSignal;
7301
+ }): PromiseLike<{
7302
+ sessionId: string;
7303
+ sdp: string;
7304
+ }>;
7191
7305
  /**
7192
7306
  * Server-side: Creates an ephemeral client secret for authenticating
7193
7307
  * browser-side WebSocket connections. The secret is short-lived and
@@ -7195,13 +7309,13 @@ type RealtimeModelV4 = {
7195
7309
  *
7196
7310
  * Naming: "do" prefix to prevent accidental direct usage by the user.
7197
7311
  */
7198
- doCreateClientSecret(options: RealtimeModelV4ClientSecretOptions): PromiseLike<RealtimeModelV4ClientSecretResult>;
7312
+ doCreateClientSecret?(options: RealtimeModelV4ClientSecretOptions): PromiseLike<RealtimeModelV4ClientSecretResult>;
7199
7313
  /**
7200
7314
  * Browser-side: Returns the WebSocket URL and subprotocols to use
7201
7315
  * when connecting. Each provider has its own authentication mechanism
7202
7316
  * (e.g. OpenAI uses subprotocol headers, xAI may use query params).
7203
7317
  */
7204
- getWebSocketConfig(options: {
7318
+ getWebSocketConfig?(options: {
7205
7319
  token: string;
7206
7320
  url: string;
7207
7321
  }): {
@@ -7218,6 +7332,8 @@ type RealtimeModelV4 = {
7218
7332
  * text, and turn-complete data in one message).
7219
7333
  */
7220
7334
  parseServerEvent(raw: unknown): RealtimeModelV4ServerEvent | RealtimeModelV4ServerEvent[];
7335
+ /** Create a raw-event parser per connection and discard it on disconnect. */
7336
+ createServerEventParser?(): (raw: unknown) => RealtimeModelV4ServerEvent | RealtimeModelV4ServerEvent[];
7221
7337
  /**
7222
7338
  * Browser-side: Serializes a normalized client event into the
7223
7339
  * provider's native JSON format for sending over the WebSocket.
@@ -8242,4 +8358,4 @@ type VideoModelV3 = {
8242
8358
  }>;
8243
8359
  };
8244
8360
 
8245
- export { AISDKError, APICallError, type EmbeddingModelV2, type EmbeddingModelV2Embedding, type EmbeddingModelV3, type EmbeddingModelV3CallOptions, type EmbeddingModelV3Embedding, type EmbeddingModelV3Middleware, type EmbeddingModelV3Result, type EmbeddingModelV4, type EmbeddingModelV4CallOptions, type EmbeddingModelV4Embedding, type EmbeddingModelV4Middleware, type EmbeddingModelV4Result, EmptyResponseBodyError, type BatchV4 as Experimental_BatchV4, type BatchV4CancelResult as Experimental_BatchV4CancelResult, type BatchV4Error as Experimental_BatchV4Error, type BatchV4ItemResult as Experimental_BatchV4ItemResult, type BatchV4ListItem as Experimental_BatchV4ListItem, type BatchV4ListOptions as Experimental_BatchV4ListOptions, type BatchV4ListResult as Experimental_BatchV4ListResult, type BatchV4ModelIds as Experimental_BatchV4ModelIds, type BatchV4OperationOptions as Experimental_BatchV4OperationOptions, type BatchV4Request as Experimental_BatchV4Request, type BatchV4RequestBase as Experimental_BatchV4RequestBase, type BatchV4StartOptions as Experimental_BatchV4StartOptions, type BatchV4StartResult as Experimental_BatchV4StartResult, type BatchV4Status as Experimental_BatchV4Status, type RealtimeFactoryV4 as Experimental_RealtimeFactoryV4, type RealtimeFactoryV4GetTokenOptions as Experimental_RealtimeFactoryV4GetTokenOptions, type RealtimeFactoryV4GetTokenResult as Experimental_RealtimeFactoryV4GetTokenResult, type RealtimeModelV4 as Experimental_RealtimeModelV4, type RealtimeModelV4AudioMessage as Experimental_RealtimeModelV4AudioMessage, type RealtimeModelV4ClientEvent as Experimental_RealtimeModelV4ClientEvent, type RealtimeModelV4ClientSecretOptions as Experimental_RealtimeModelV4ClientSecretOptions, type RealtimeModelV4ClientSecretResult as Experimental_RealtimeModelV4ClientSecretResult, type RealtimeModelV4ConversationItem as Experimental_RealtimeModelV4ConversationItem, type RealtimeModelV4FunctionCallOutput as Experimental_RealtimeModelV4FunctionCallOutput, type RealtimeModelV4ServerEvent as Experimental_RealtimeModelV4ServerEvent, type RealtimeModelV4SessionConfig as Experimental_RealtimeModelV4SessionConfig, type RealtimeModelV4TextMessage as Experimental_RealtimeModelV4TextMessage, type RealtimeModelV4ToolDefinition as Experimental_RealtimeModelV4ToolDefinition, type SpeechTranslationModelV4 as Experimental_SpeechTranslationModelV4, type SpeechTranslationModelV4StreamOptions as Experimental_SpeechTranslationModelV4StreamOptions, type SpeechTranslationModelV4StreamPart as Experimental_SpeechTranslationModelV4StreamPart, type SpeechTranslationModelV4StreamResult as Experimental_SpeechTranslationModelV4StreamResult, type SpeechTranslationModelV4Usage as Experimental_SpeechTranslationModelV4Usage, type TextBatchV4ItemResult as Experimental_TextBatchV4ItemResult, type TextBatchV4Request as Experimental_TextBatchV4Request, type TranscriptionModelV4StreamOptions as Experimental_TranscriptionModelV4StreamOptions, type TranscriptionModelV4StreamPart as Experimental_TranscriptionModelV4StreamPart, type TranscriptionModelV4StreamResult as Experimental_TranscriptionModelV4StreamResult, type VideoModelV3 as Experimental_VideoModelV3, type VideoModelV3CallOptions as Experimental_VideoModelV3CallOptions, type VideoModelV3File as Experimental_VideoModelV3File, type VideoModelV3FrameImage as Experimental_VideoModelV3FrameImage, type VideoModelV3FrameType as Experimental_VideoModelV3FrameType, type VideoModelV3VideoData as Experimental_VideoModelV3VideoData, type VideoModelV4 as Experimental_VideoModelV4, type VideoModelV4CallOptions as Experimental_VideoModelV4CallOptions, type VideoModelV4File as Experimental_VideoModelV4File, type VideoModelV4FrameImage as Experimental_VideoModelV4FrameImage, type VideoModelV4FrameType as Experimental_VideoModelV4FrameType, type VideoModelV4OperationStartResult as Experimental_VideoModelV4OperationStartResult, type VideoModelV4OperationStatusResult as Experimental_VideoModelV4OperationStatusResult, type VideoModelV4OperationWebhook as Experimental_VideoModelV4OperationWebhook, type VideoModelV4Result as Experimental_VideoModelV4Result, type VideoModelV4VideoData as Experimental_VideoModelV4VideoData, type FilesV4, type FilesV4DeleteFileCallOptions, type FilesV4DeleteFileResult, type FilesV4DownloadFileCallOptions, type FilesV4DownloadFileResult, type FilesV4GetFileMetadataCallOptions, type FilesV4GetFileMetadataResult, type FilesV4UploadFileCallOptions, type FilesV4UploadFileResult, type FilesV4UploadFileStreamData, type ImageModelV2, type ImageModelV2CallOptions, type ImageModelV2CallWarning, type ImageModelV2ProviderMetadata, type ImageModelV3, type ImageModelV3CallOptions, type ImageModelV3File, type ImageModelV3Middleware, type ImageModelV3ProviderMetadata, type ImageModelV3Usage, type ImageModelV4, type ImageModelV4CallOptions, type ImageModelV4File, type ImageModelV4Middleware, type ImageModelV4ProviderMetadata, type ImageModelV4Result, type ImageModelV4Usage, InvalidArgumentError, InvalidPromptError, InvalidResponseDataError, type JSONArray, type JSONObject, JSONParseError, type JSONValue, type LanguageModelV2, type LanguageModelV2CallOptions, type LanguageModelV2CallWarning, type LanguageModelV2Content, type LanguageModelV2DataContent, type LanguageModelV2File, type LanguageModelV2FilePart, type LanguageModelV2FinishReason, type LanguageModelV2FunctionTool, type LanguageModelV2Message, type LanguageModelV2Middleware, type LanguageModelV2Prompt, type LanguageModelV2ProviderDefinedTool, type LanguageModelV2Reasoning, type LanguageModelV2ReasoningPart, type LanguageModelV2ResponseMetadata, type LanguageModelV2Source, type LanguageModelV2StreamPart, type LanguageModelV2Text, type LanguageModelV2TextPart, type LanguageModelV2ToolCall, type LanguageModelV2ToolCallPart, type LanguageModelV2ToolChoice, type LanguageModelV2ToolResultOutput, type LanguageModelV2ToolResultPart, type LanguageModelV2Usage, type LanguageModelV3, type LanguageModelV3CallOptions, type LanguageModelV3Content, type LanguageModelV3DataContent, type LanguageModelV3File, type LanguageModelV3FilePart, type LanguageModelV3FinishReason, type LanguageModelV3FunctionTool, type LanguageModelV3GenerateResult, type LanguageModelV3Message, type LanguageModelV3Middleware, type LanguageModelV3Prompt, type LanguageModelV3ProviderTool, type LanguageModelV3Reasoning, type LanguageModelV3ReasoningPart, type LanguageModelV3ResponseMetadata, type LanguageModelV3Source, type LanguageModelV3StreamPart, type LanguageModelV3StreamResult, type LanguageModelV3Text, type LanguageModelV3TextPart, type LanguageModelV3ToolApprovalRequest, type LanguageModelV3ToolApprovalResponsePart, type LanguageModelV3ToolCall, type LanguageModelV3ToolCallPart, type LanguageModelV3ToolChoice, type LanguageModelV3ToolResult, type LanguageModelV3ToolResultOutput, type LanguageModelV3ToolResultPart, type LanguageModelV3Usage, type LanguageModelV4, type LanguageModelV4CallOptions, type LanguageModelV4Content, type LanguageModelV4CustomContent, type LanguageModelV4CustomPart, type LanguageModelV4File, type LanguageModelV4FilePart, type LanguageModelV4FinishReason, type LanguageModelV4FunctionTool, type LanguageModelV4GenerateResult, type LanguageModelV4Message, type LanguageModelV4Middleware, type LanguageModelV4Prompt, type LanguageModelV4ProviderTool, type LanguageModelV4Reasoning, type LanguageModelV4ReasoningFile, type LanguageModelV4ReasoningFilePart, type LanguageModelV4ReasoningPart, type LanguageModelV4ResponseMetadata, type LanguageModelV4Source, type LanguageModelV4StreamPart, type LanguageModelV4StreamResult, type LanguageModelV4Text, type LanguageModelV4TextPart, type LanguageModelV4ToolApprovalRequest, type LanguageModelV4ToolApprovalResponsePart, type LanguageModelV4ToolCall, type LanguageModelV4ToolCallPart, type LanguageModelV4ToolChoice, type LanguageModelV4ToolResult, type LanguageModelV4ToolResultOutput, type LanguageModelV4ToolResultPart, type LanguageModelV4Usage, LoadAPIKeyError, LoadSettingError, NoContentGeneratedError, NoSuchModelError, NoSuchProviderReferenceError, type ProviderV2, type ProviderV3, type ProviderV4, type RerankingModelV3, type RerankingModelV3CallOptions, type RerankingModelV4, type RerankingModelV4CallOptions, type RerankingModelV4Result, type SharedV2Headers, type SharedV2ProviderMetadata, type SharedV2ProviderOptions, type SharedV3Headers, type SharedV3ProviderMetadata, type SharedV3ProviderOptions, type SharedV3Warning, type SharedV4AudioFormat, type SharedV4FileData, type SharedV4FileDataData, type SharedV4FileDataReference, type SharedV4FileDataText, type SharedV4FileDataUrl, type SharedV4Headers, type SharedV4ProviderMetadata, type SharedV4ProviderOptions, type SharedV4ProviderReference, type SharedV4Warning, type SkillsV4, type SkillsV4File, type SkillsV4UploadSkillCallOptions, type SkillsV4UploadSkillResult, type SpeechModelV2, type SpeechModelV2CallOptions, type SpeechModelV2CallWarning, type SpeechModelV3, type SpeechModelV3CallOptions, type SpeechModelV4, type SpeechModelV4CallOptions, type SpeechModelV4Result, TooManyEmbeddingValuesForCallError, type TranscriptionModelV2, type TranscriptionModelV2CallOptions, type TranscriptionModelV2CallWarning, type TranscriptionModelV3, type TranscriptionModelV3CallOptions, type TranscriptionModelV4, type TranscriptionModelV4CallOptions, type TranscriptionModelV4Result, type TypeValidationContext, TypeValidationError, UnsupportedFunctionalityError, getErrorMessage, isJSONArray, isJSONObject, isJSONValue };
8361
+ export { AISDKError, APICallError, type EmbeddingModelV2, type EmbeddingModelV2Embedding, type EmbeddingModelV3, type EmbeddingModelV3CallOptions, type EmbeddingModelV3Embedding, type EmbeddingModelV3Middleware, type EmbeddingModelV3Result, type EmbeddingModelV4, type EmbeddingModelV4CallOptions, type EmbeddingModelV4Embedding, type EmbeddingModelV4Middleware, type EmbeddingModelV4Result, EmptyResponseBodyError, type BatchV4 as Experimental_BatchV4, type BatchV4CancelResult as Experimental_BatchV4CancelResult, type BatchV4Error as Experimental_BatchV4Error, type BatchV4ItemResult as Experimental_BatchV4ItemResult, type BatchV4ListItem as Experimental_BatchV4ListItem, type BatchV4ListOptions as Experimental_BatchV4ListOptions, type BatchV4ListResult as Experimental_BatchV4ListResult, type BatchV4ModelIds as Experimental_BatchV4ModelIds, type BatchV4OperationOptions as Experimental_BatchV4OperationOptions, type BatchV4Request as Experimental_BatchV4Request, type BatchV4RequestBase as Experimental_BatchV4RequestBase, type BatchV4StartOptions as Experimental_BatchV4StartOptions, type BatchV4StartResult as Experimental_BatchV4StartResult, type BatchV4Status as Experimental_BatchV4Status, type ImageBatchV4ItemResult as Experimental_ImageBatchV4ItemResult, type ImageBatchV4Request as Experimental_ImageBatchV4Request, type RealtimeFactoryV4 as Experimental_RealtimeFactoryV4, type RealtimeFactoryV4GetTokenOptions as Experimental_RealtimeFactoryV4GetTokenOptions, type RealtimeFactoryV4GetTokenResult as Experimental_RealtimeFactoryV4GetTokenResult, type RealtimeModelV4 as Experimental_RealtimeModelV4, type RealtimeModelV4AudioMessage as Experimental_RealtimeModelV4AudioMessage, type RealtimeModelV4ClientEvent as Experimental_RealtimeModelV4ClientEvent, type RealtimeModelV4ClientSecretOptions as Experimental_RealtimeModelV4ClientSecretOptions, type RealtimeModelV4ClientSecretResult as Experimental_RealtimeModelV4ClientSecretResult, type RealtimeModelV4ConversationItem as Experimental_RealtimeModelV4ConversationItem, type RealtimeModelV4FunctionCallOutput as Experimental_RealtimeModelV4FunctionCallOutput, type RealtimeModelV4ServerEvent as Experimental_RealtimeModelV4ServerEvent, type RealtimeModelV4SessionConfig as Experimental_RealtimeModelV4SessionConfig, type RealtimeModelV4TextMessage as Experimental_RealtimeModelV4TextMessage, type RealtimeModelV4ToolDefinition as Experimental_RealtimeModelV4ToolDefinition, type SpeechTranslationModelV4 as Experimental_SpeechTranslationModelV4, type SpeechTranslationModelV4StreamOptions as Experimental_SpeechTranslationModelV4StreamOptions, type SpeechTranslationModelV4StreamPart as Experimental_SpeechTranslationModelV4StreamPart, type SpeechTranslationModelV4StreamResult as Experimental_SpeechTranslationModelV4StreamResult, type SpeechTranslationModelV4Usage as Experimental_SpeechTranslationModelV4Usage, type TextBatchV4ItemResult as Experimental_TextBatchV4ItemResult, type TextBatchV4Request as Experimental_TextBatchV4Request, type TranscriptionModelV4StreamOptions as Experimental_TranscriptionModelV4StreamOptions, type TranscriptionModelV4StreamPart as Experimental_TranscriptionModelV4StreamPart, type TranscriptionModelV4StreamResult as Experimental_TranscriptionModelV4StreamResult, type VideoModelV3 as Experimental_VideoModelV3, type VideoModelV3CallOptions as Experimental_VideoModelV3CallOptions, type VideoModelV3File as Experimental_VideoModelV3File, type VideoModelV3FrameImage as Experimental_VideoModelV3FrameImage, type VideoModelV3FrameType as Experimental_VideoModelV3FrameType, type VideoModelV3VideoData as Experimental_VideoModelV3VideoData, type VideoModelV4 as Experimental_VideoModelV4, type VideoModelV4CallOptions as Experimental_VideoModelV4CallOptions, type VideoModelV4File as Experimental_VideoModelV4File, type VideoModelV4FrameImage as Experimental_VideoModelV4FrameImage, type VideoModelV4FrameType as Experimental_VideoModelV4FrameType, type VideoModelV4OperationStartResult as Experimental_VideoModelV4OperationStartResult, type VideoModelV4OperationStatusResult as Experimental_VideoModelV4OperationStatusResult, type VideoModelV4OperationWebhook as Experimental_VideoModelV4OperationWebhook, type VideoModelV4Result as Experimental_VideoModelV4Result, type VideoModelV4VideoData as Experimental_VideoModelV4VideoData, type FilesV4, type FilesV4DeleteFileCallOptions, type FilesV4DeleteFileResult, type FilesV4DownloadFileCallOptions, type FilesV4DownloadFileResult, type FilesV4GetFileMetadataCallOptions, type FilesV4GetFileMetadataResult, type FilesV4UploadFileCallOptions, type FilesV4UploadFileResult, type FilesV4UploadFileStreamData, type ImageModelV2, type ImageModelV2CallOptions, type ImageModelV2CallWarning, type ImageModelV2ProviderMetadata, type ImageModelV3, type ImageModelV3CallOptions, type ImageModelV3File, type ImageModelV3Middleware, type ImageModelV3ProviderMetadata, type ImageModelV3Usage, type ImageModelV4, type ImageModelV4CallOptions, type ImageModelV4File, type ImageModelV4Middleware, type ImageModelV4ProviderMetadata, type ImageModelV4Result, type ImageModelV4Usage, InvalidArgumentError, InvalidPromptError, InvalidResponseDataError, type JSONArray, type JSONObject, JSONParseError, type JSONValue, type LanguageModelV2, type LanguageModelV2CallOptions, type LanguageModelV2CallWarning, type LanguageModelV2Content, type LanguageModelV2DataContent, type LanguageModelV2File, type LanguageModelV2FilePart, type LanguageModelV2FinishReason, type LanguageModelV2FunctionTool, type LanguageModelV2Message, type LanguageModelV2Middleware, type LanguageModelV2Prompt, type LanguageModelV2ProviderDefinedTool, type LanguageModelV2Reasoning, type LanguageModelV2ReasoningPart, type LanguageModelV2ResponseMetadata, type LanguageModelV2Source, type LanguageModelV2StreamPart, type LanguageModelV2Text, type LanguageModelV2TextPart, type LanguageModelV2ToolCall, type LanguageModelV2ToolCallPart, type LanguageModelV2ToolChoice, type LanguageModelV2ToolResultOutput, type LanguageModelV2ToolResultPart, type LanguageModelV2Usage, type LanguageModelV3, type LanguageModelV3CallOptions, type LanguageModelV3Content, type LanguageModelV3DataContent, type LanguageModelV3File, type LanguageModelV3FilePart, type LanguageModelV3FinishReason, type LanguageModelV3FunctionTool, type LanguageModelV3GenerateResult, type LanguageModelV3Message, type LanguageModelV3Middleware, type LanguageModelV3Prompt, type LanguageModelV3ProviderTool, type LanguageModelV3Reasoning, type LanguageModelV3ReasoningPart, type LanguageModelV3ResponseMetadata, type LanguageModelV3Source, type LanguageModelV3StreamPart, type LanguageModelV3StreamResult, type LanguageModelV3Text, type LanguageModelV3TextPart, type LanguageModelV3ToolApprovalRequest, type LanguageModelV3ToolApprovalResponsePart, type LanguageModelV3ToolCall, type LanguageModelV3ToolCallPart, type LanguageModelV3ToolChoice, type LanguageModelV3ToolResult, type LanguageModelV3ToolResultOutput, type LanguageModelV3ToolResultPart, type LanguageModelV3Usage, type LanguageModelV4, type LanguageModelV4CallOptions, type LanguageModelV4Content, type LanguageModelV4CustomContent, type LanguageModelV4CustomPart, type LanguageModelV4File, type LanguageModelV4FilePart, type LanguageModelV4FinishReason, type LanguageModelV4FunctionTool, type LanguageModelV4GenerateResult, type LanguageModelV4Message, type LanguageModelV4Middleware, type LanguageModelV4Prompt, type LanguageModelV4ProviderTool, type LanguageModelV4Reasoning, type LanguageModelV4ReasoningFile, type LanguageModelV4ReasoningFilePart, type LanguageModelV4ReasoningPart, type LanguageModelV4ResponseMetadata, type LanguageModelV4Source, type LanguageModelV4StreamPart, type LanguageModelV4StreamResult, type LanguageModelV4Text, type LanguageModelV4TextPart, type LanguageModelV4ToolApprovalRequest, type LanguageModelV4ToolApprovalResponsePart, type LanguageModelV4ToolCall, type LanguageModelV4ToolCallPart, type LanguageModelV4ToolChoice, type LanguageModelV4ToolResult, type LanguageModelV4ToolResultOutput, type LanguageModelV4ToolResultPart, type LanguageModelV4Usage, LoadAPIKeyError, LoadSettingError, NoContentGeneratedError, NoSuchModelError, NoSuchProviderReferenceError, type ProviderV2, type ProviderV3, type ProviderV4, type RerankingModelV3, type RerankingModelV3CallOptions, type RerankingModelV4, type RerankingModelV4CallOptions, type RerankingModelV4Result, type SharedV2Headers, type SharedV2ProviderMetadata, type SharedV2ProviderOptions, type SharedV3Headers, type SharedV3ProviderMetadata, type SharedV3ProviderOptions, type SharedV3Warning, type SharedV4AudioFormat, type SharedV4FileData, type SharedV4FileDataData, type SharedV4FileDataReference, type SharedV4FileDataText, type SharedV4FileDataUrl, type SharedV4Headers, type SharedV4ProviderMetadata, type SharedV4ProviderOptions, type SharedV4ProviderReference, type SharedV4Warning, type SkillsV4, type SkillsV4File, type SkillsV4UploadSkillCallOptions, type SkillsV4UploadSkillResult, type SpeechModelV2, type SpeechModelV2CallOptions, type SpeechModelV2CallWarning, type SpeechModelV3, type SpeechModelV3CallOptions, type SpeechModelV4, type SpeechModelV4CallOptions, type SpeechModelV4Result, TooManyEmbeddingValuesForCallError, type TranscriptionModelV2, type TranscriptionModelV2CallOptions, type TranscriptionModelV2CallWarning, type TranscriptionModelV3, type TranscriptionModelV3CallOptions, type TranscriptionModelV4, type TranscriptionModelV4CallOptions, type TranscriptionModelV4Result, type TypeValidationContext, TypeValidationError, UnsupportedFunctionalityError, getErrorMessage, isJSONArray, isJSONObject, isJSONValue };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/provider",
3
- "version": "4.0.13",
3
+ "version": "4.0.15",
4
4
  "type": "module",
5
5
  "license": "Apache-2.0",
6
6
  "sideEffects": false,
@@ -4,6 +4,8 @@ import type {
4
4
  SharedV4Warning,
5
5
  } from '../../shared';
6
6
  import type { LanguageModelV4GenerateResult } from '../../language-model/v4/language-model-v4-generate-result';
7
+ import type { ImageModelV4Result } from '../../image-model/v4';
8
+ import type { ImageBatchV4Request } from './image-batch-v4-request';
7
9
  import type { TextBatchV4Request } from './text-batch-v4-request';
8
10
 
9
11
  /**
@@ -12,7 +14,8 @@ import type { TextBatchV4Request } from './text-batch-v4-request';
12
14
  * Additional modality-specific model ID types can be added to this mapping.
13
15
  */
14
16
  export type BatchV4ModelIds = {
15
- readonly text: string;
17
+ readonly text?: string;
18
+ readonly image?: string;
16
19
  };
17
20
 
18
21
  type BatchV4CallOptions = {
@@ -59,7 +62,8 @@ export type BatchV4Status = {
59
62
  };
60
63
 
61
64
  export type BatchV4Request<ModelIds extends BatchV4ModelIds = BatchV4ModelIds> =
62
- TextBatchV4Request<ModelIds['text']>;
65
+ | TextBatchV4Request<ModelIds['text'] & string>
66
+ | ImageBatchV4Request<ModelIds['image'] & string>;
63
67
 
64
68
  /**
65
69
  * Options for starting a batch of requests discriminated by modality.
@@ -148,13 +152,20 @@ export type TextBatchV4ItemResult = {
148
152
  readonly type: 'text';
149
153
  } & BatchV4ItemResultBase<LanguageModelV4GenerateResult>;
150
154
 
155
+ /**
156
+ * A complete terminal result for one request in an image batch.
157
+ */
158
+ export type ImageBatchV4ItemResult = {
159
+ readonly type: 'image';
160
+ } & BatchV4ItemResultBase<ImageModelV4Result>;
161
+
151
162
  /**
152
163
  * A complete terminal result for one request in a batch, discriminated by
153
164
  * modality.
154
165
  *
155
166
  * Additional modality-specific item results can be added to this union.
156
167
  */
157
- export type BatchV4ItemResult = TextBatchV4ItemResult;
168
+ export type BatchV4ItemResult = TextBatchV4ItemResult | ImageBatchV4ItemResult;
158
169
 
159
170
  /**
160
171
  * Specification for a batch interface that implements batch interface version 4.
@@ -0,0 +1,21 @@
1
+ import type { ImageModelV4CallOptions } from '../../image-model/v4';
2
+ import type { BatchV4RequestBase } from './batch-v4-request';
3
+
4
+ /**
5
+ * One image generation request in a batch.
6
+ */
7
+ export type ImageBatchV4Request<ModelId extends string = string> =
8
+ BatchV4RequestBase<ModelId> & {
9
+ readonly type: 'image';
10
+ readonly options: Pick<
11
+ ImageModelV4CallOptions,
12
+ | 'prompt'
13
+ | 'n'
14
+ | 'size'
15
+ | 'aspectRatio'
16
+ | 'seed'
17
+ | 'files'
18
+ | 'mask'
19
+ | 'providerOptions'
20
+ >;
21
+ };
@@ -12,7 +12,9 @@ export type {
12
12
  BatchV4StartOptions as Experimental_BatchV4StartOptions,
13
13
  BatchV4StartResult as Experimental_BatchV4StartResult,
14
14
  BatchV4Status as Experimental_BatchV4Status,
15
+ ImageBatchV4ItemResult as Experimental_ImageBatchV4ItemResult,
15
16
  TextBatchV4ItemResult as Experimental_TextBatchV4ItemResult,
16
17
  } from './batch-v4';
17
18
  export type { BatchV4RequestBase as Experimental_BatchV4RequestBase } from './batch-v4-request';
18
19
  export type { TextBatchV4Request as Experimental_TextBatchV4Request } from './text-batch-v4-request';
20
+ export type { ImageBatchV4Request as Experimental_ImageBatchV4Request } from './image-batch-v4-request';
@@ -1,3 +1,4 @@
1
+ import type { SharedV4ProviderOptions } from '../../shared/v4/shared-v4-provider-options';
1
2
  import type { RealtimeModelV4ConversationItem } from './realtime-model-v4-conversation-item';
2
3
  import type { RealtimeModelV4SessionConfig } from './realtime-model-v4-session-config';
3
4
 
@@ -12,6 +13,31 @@ export type RealtimeModelV4ClientEvent =
12
13
  | {
13
14
  type: 'session-update';
14
15
  config: RealtimeModelV4SessionConfig;
16
+ eventId?: string;
17
+ }
18
+ | {
19
+ type: 'session-start';
20
+ config: RealtimeModelV4SessionConfig;
21
+ eventId?: string;
22
+ }
23
+ | {
24
+ type: 'session-close';
25
+ eventId?: string;
26
+ }
27
+ | {
28
+ type: 'input-audio-mute';
29
+ eventId?: string;
30
+ }
31
+ | {
32
+ type: 'input-audio-unmute';
33
+ eventId?: string;
34
+ }
35
+ | {
36
+ type: 'context-append';
37
+ content: string;
38
+ delegationId: string | null;
39
+ eventId?: string;
40
+ providerOptions?: SharedV4ProviderOptions;
15
41
  }
16
42
 
17
43
  // ── Input audio buffer ─────────────────────────────────────────────
@@ -22,6 +48,7 @@ export type RealtimeModelV4ClientEvent =
22
48
  * Base64-encoded audio chunk to append to the input buffer.
23
49
  */
24
50
  audio: string;
51
+ eventId?: string;
25
52
  }
26
53
  | {
27
54
  type: 'input-audio-commit';
@@ -6,8 +6,55 @@
6
6
  * event data for debugging and provider-specific access.
7
7
  */
8
8
  export type RealtimeModelV4ServerEvent =
9
+ | {
10
+ type: 'session-started';
11
+ sessionId: string;
12
+ delegationMode?: 'client' | 'provider';
13
+ raw: unknown;
14
+ }
15
+ | {
16
+ type: 'session-closed';
17
+ sessionId?: string;
18
+ usage: { seconds: number };
19
+ reason: string;
20
+ raw: unknown;
21
+ }
22
+ | {
23
+ type: 'session-usage';
24
+ /** Cumulative duration snapshot, not an increment. */
25
+ usage: { seconds: number };
26
+ contextWindowUsageRatio?: number;
27
+ raw: unknown;
28
+ }
29
+ | {
30
+ type: 'audio-chunk';
31
+ delta: string;
32
+ raw: unknown;
33
+ }
34
+ | {
35
+ type: 'transcript-fragment';
36
+ speaker: 'user' | 'assistant';
37
+ delta: string;
38
+ startMs: number;
39
+ endMs: number;
40
+ raw: unknown;
41
+ }
42
+ | {
43
+ type: 'delegation-created';
44
+ delegationId: string;
45
+ target?: 'client' | 'provider';
46
+ offsetMs?: number;
47
+ responseId?: string;
48
+ raw: unknown;
49
+ }
50
+ | {
51
+ type: 'command-acknowledged';
52
+ /** Provider-native command name that was acknowledged. */
53
+ command: string;
54
+ clientEventId?: string;
55
+ raw: unknown;
56
+ }
9
57
  // ── Session lifecycle ──────────────────────────────────────────────
10
-
11
58
  | {
12
59
  type: 'session-created';
13
60
  sessionId?: string;
@@ -184,6 +231,7 @@ export type RealtimeModelV4ServerEvent =
184
231
  type: 'error';
185
232
  message: string;
186
233
  code?: string;
234
+ clientEventId?: string;
187
235
  raw: unknown;
188
236
  }
189
237
 
@@ -29,6 +29,37 @@ export type RealtimeModelV4 = {
29
29
  */
30
30
  readonly modelId: string;
31
31
 
32
+ /** Conversation semantics and supported transports, when declared. */
33
+ readonly capabilities?: {
34
+ conversation: 'continuous' | 'turn-based';
35
+ transports: readonly ('websocket' | 'webrtc')[];
36
+ /** Omission preserves the legacy client-secret WebSocket connection. */
37
+ connections?: readonly (
38
+ | 'client-secret-websocket'
39
+ | 'server-websocket'
40
+ | 'webrtc'
41
+ )[];
42
+ /** Omission preserves session-update startup. WebRTC setup may start the session. */
43
+ startup?: 'session-start' | 'session-update';
44
+ /** Omission preserves transport-close finalization. */
45
+ finalization?: 'session-close' | 'transport-close';
46
+ };
47
+
48
+ /** Server-only connection settings. Headers can contain long-lived credentials. */
49
+ getServerWebSocketConfig?():
50
+ | { url: string; headers: Record<string, string> }
51
+ | PromiseLike<{ url: string; headers: Record<string, string> }>;
52
+
53
+ /** Provider-specific setup for the WebRTC event channel. */
54
+ getWebRTCConfig?(): { dataChannelLabel: string };
55
+
56
+ /** Server-side SDP exchange. Return only the answer and session ID to the client. */
57
+ doCreateWebRTCSession?(options: {
58
+ sdp: string;
59
+ sessionConfig?: RealtimeModelV4SessionConfig;
60
+ abortSignal?: AbortSignal;
61
+ }): PromiseLike<{ sessionId: string; sdp: string }>;
62
+
32
63
  /**
33
64
  * Server-side: Creates an ephemeral client secret for authenticating
34
65
  * browser-side WebSocket connections. The secret is short-lived and
@@ -36,7 +67,7 @@ export type RealtimeModelV4 = {
36
67
  *
37
68
  * Naming: "do" prefix to prevent accidental direct usage by the user.
38
69
  */
39
- doCreateClientSecret(
70
+ doCreateClientSecret?(
40
71
  options: RealtimeModelV4ClientSecretOptions,
41
72
  ): PromiseLike<RealtimeModelV4ClientSecretResult>;
42
73
 
@@ -45,7 +76,7 @@ export type RealtimeModelV4 = {
45
76
  * when connecting. Each provider has its own authentication mechanism
46
77
  * (e.g. OpenAI uses subprotocol headers, xAI may use query params).
47
78
  */
48
- getWebSocketConfig(options: { token: string; url: string }): {
79
+ getWebSocketConfig?(options: { token: string; url: string }): {
49
80
  url: string;
50
81
  protocols?: string[];
51
82
  };
@@ -63,6 +94,11 @@ export type RealtimeModelV4 = {
63
94
  raw: unknown,
64
95
  ): RealtimeModelV4ServerEvent | RealtimeModelV4ServerEvent[];
65
96
 
97
+ /** Create a raw-event parser per connection and discard it on disconnect. */
98
+ createServerEventParser?(): (
99
+ raw: unknown,
100
+ ) => RealtimeModelV4ServerEvent | RealtimeModelV4ServerEvent[];
101
+
66
102
  /**
67
103
  * Browser-side: Serializes a normalized client event into the
68
104
  * provider's native JSON format for sending over the WebSocket.