@tanstack/ai 0.0.2 → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +26 -0
  2. package/dist/esm/activities/chat/adapter.d.ts +100 -0
  3. package/dist/esm/activities/chat/adapter.js +14 -0
  4. package/dist/esm/activities/chat/adapter.js.map +1 -0
  5. package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.d.ts +4 -4
  6. package/dist/esm/activities/chat/agent-loop-strategies.js.map +1 -0
  7. package/dist/esm/activities/chat/index.d.ts +165 -0
  8. package/dist/esm/{core/chat.js → activities/chat/index.js} +131 -33
  9. package/dist/esm/activities/chat/index.js.map +1 -0
  10. package/dist/esm/{message-converters.d.ts → activities/chat/messages.d.ts} +1 -1
  11. package/dist/esm/{message-converters.js → activities/chat/messages.js} +7 -7
  12. package/dist/esm/activities/chat/messages.js.map +1 -0
  13. package/dist/esm/activities/chat/stream/json-parser.js.map +1 -0
  14. package/dist/esm/{stream → activities/chat/stream}/message-updaters.d.ts +1 -1
  15. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -0
  16. package/dist/esm/{stream → activities/chat/stream}/processor.d.ts +1 -1
  17. package/dist/esm/{stream → activities/chat/stream}/processor.js +1 -1
  18. package/dist/esm/activities/chat/stream/processor.js.map +1 -0
  19. package/dist/esm/activities/chat/stream/strategies.js.map +1 -0
  20. package/dist/esm/{stream → activities/chat/stream}/types.d.ts +2 -9
  21. package/dist/esm/{tools → activities/chat/tools}/tool-calls.d.ts +1 -1
  22. package/dist/esm/{tools → activities/chat/tools}/tool-calls.js +9 -5
  23. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -0
  24. package/dist/esm/{tools → activities/chat/tools}/tool-definition.d.ts +14 -14
  25. package/dist/esm/activities/chat/tools/tool-definition.js.map +1 -0
  26. package/dist/esm/activities/chat/tools/zod-converter.d.ts +69 -0
  27. package/dist/esm/activities/chat/tools/zod-converter.js +99 -0
  28. package/dist/esm/activities/chat/tools/zod-converter.js.map +1 -0
  29. package/dist/esm/activities/generateImage/adapter.d.ts +68 -0
  30. package/dist/esm/activities/generateImage/adapter.js +14 -0
  31. package/dist/esm/activities/generateImage/adapter.js.map +1 -0
  32. package/dist/esm/activities/generateImage/index.d.ts +89 -0
  33. package/dist/esm/activities/generateImage/index.js +15 -0
  34. package/dist/esm/activities/generateImage/index.js.map +1 -0
  35. package/dist/esm/activities/generateSpeech/adapter.d.ts +62 -0
  36. package/dist/esm/activities/generateSpeech/adapter.js +14 -0
  37. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -0
  38. package/dist/esm/activities/generateSpeech/index.d.ts +69 -0
  39. package/dist/esm/activities/generateSpeech/index.js +15 -0
  40. package/dist/esm/activities/generateSpeech/index.js.map +1 -0
  41. package/dist/esm/activities/generateTranscription/adapter.d.ts +62 -0
  42. package/dist/esm/activities/generateTranscription/adapter.js +14 -0
  43. package/dist/esm/activities/generateTranscription/adapter.js.map +1 -0
  44. package/dist/esm/activities/generateTranscription/index.d.ts +71 -0
  45. package/dist/esm/activities/generateTranscription/index.js +15 -0
  46. package/dist/esm/activities/generateTranscription/index.js.map +1 -0
  47. package/dist/esm/activities/generateVideo/adapter.d.ts +80 -0
  48. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  49. package/dist/esm/activities/generateVideo/adapter.js.map +1 -0
  50. package/dist/esm/activities/generateVideo/index.d.ts +136 -0
  51. package/dist/esm/activities/generateVideo/index.js +47 -0
  52. package/dist/esm/activities/generateVideo/index.js.map +1 -0
  53. package/dist/esm/activities/index.d.ts +22 -0
  54. package/dist/esm/activities/index.js +34 -0
  55. package/dist/esm/activities/index.js.map +1 -0
  56. package/dist/esm/activities/summarize/adapter.d.ts +74 -0
  57. package/dist/esm/activities/summarize/adapter.js +14 -0
  58. package/dist/esm/activities/summarize/adapter.js.map +1 -0
  59. package/dist/esm/activities/summarize/index.d.ts +100 -0
  60. package/dist/esm/activities/summarize/index.js +90 -0
  61. package/dist/esm/activities/summarize/index.js.map +1 -0
  62. package/dist/esm/event-client.d.ts +4 -45
  63. package/dist/esm/event-client.js +0 -49
  64. package/dist/esm/event-client.js.map +1 -1
  65. package/dist/esm/index.d.ts +16 -14
  66. package/dist/esm/index.js +28 -19
  67. package/dist/esm/stream-to-response.d.ts +95 -0
  68. package/dist/esm/stream-to-response.js +118 -0
  69. package/dist/esm/stream-to-response.js.map +1 -0
  70. package/dist/esm/types.d.ts +347 -129
  71. package/package.json +6 -2
  72. package/src/activities/chat/adapter.ts +150 -0
  73. package/src/{utilities → activities/chat}/agent-loop-strategies.ts +4 -4
  74. package/src/{core/chat.ts → activities/chat/index.ts} +427 -79
  75. package/src/{message-converters.ts → activities/chat/messages.ts} +10 -13
  76. package/src/{stream → activities/chat/stream}/message-updaters.ts +1 -1
  77. package/src/{stream → activities/chat/stream}/processor.ts +2 -5
  78. package/src/{stream → activities/chat/stream}/types.ts +8 -18
  79. package/src/{tools → activities/chat/tools}/tool-calls.ts +36 -11
  80. package/src/{tools → activities/chat/tools}/tool-definition.ts +36 -27
  81. package/src/activities/chat/tools/zod-converter.ts +235 -0
  82. package/src/activities/generateImage/adapter.ts +104 -0
  83. package/src/activities/generateImage/index.ts +162 -0
  84. package/src/activities/generateSpeech/adapter.ts +87 -0
  85. package/src/activities/generateSpeech/index.ts +122 -0
  86. package/src/activities/generateTranscription/adapter.ts +89 -0
  87. package/src/activities/generateTranscription/index.ts +132 -0
  88. package/src/activities/generateVideo/adapter.ts +116 -0
  89. package/src/activities/generateVideo/index.ts +261 -0
  90. package/src/activities/index.ts +164 -0
  91. package/src/activities/summarize/adapter.ts +107 -0
  92. package/src/activities/summarize/index.ts +287 -0
  93. package/src/event-client.ts +5 -101
  94. package/src/index.ts +58 -15
  95. package/src/stream-to-response.ts +237 -0
  96. package/src/types.ts +404 -280
  97. package/dist/esm/base-adapter.d.ts +0 -36
  98. package/dist/esm/base-adapter.js +0 -12
  99. package/dist/esm/base-adapter.js.map +0 -1
  100. package/dist/esm/core/chat-common-options.d.ts +0 -52
  101. package/dist/esm/core/chat.d.ts +0 -30
  102. package/dist/esm/core/chat.js.map +0 -1
  103. package/dist/esm/core/embedding.d.ts +0 -8
  104. package/dist/esm/core/embedding.js +0 -33
  105. package/dist/esm/core/embedding.js.map +0 -1
  106. package/dist/esm/core/summarize.d.ts +0 -9
  107. package/dist/esm/core/summarize.js +0 -36
  108. package/dist/esm/core/summarize.js.map +0 -1
  109. package/dist/esm/message-converters.js.map +0 -1
  110. package/dist/esm/stream/json-parser.js.map +0 -1
  111. package/dist/esm/stream/message-updaters.js.map +0 -1
  112. package/dist/esm/stream/processor.js.map +0 -1
  113. package/dist/esm/stream/strategies.js.map +0 -1
  114. package/dist/esm/tools/tool-calls.js.map +0 -1
  115. package/dist/esm/tools/tool-definition.js.map +0 -1
  116. package/dist/esm/tools/zod-converter.d.ts +0 -30
  117. package/dist/esm/tools/zod-converter.js +0 -36
  118. package/dist/esm/tools/zod-converter.js.map +0 -1
  119. package/dist/esm/utilities/agent-loop-strategies.js.map +0 -1
  120. package/dist/esm/utilities/chat-options.d.ts +0 -6
  121. package/dist/esm/utilities/chat-options.js +0 -7
  122. package/dist/esm/utilities/chat-options.js.map +0 -1
  123. package/dist/esm/utilities/messages.d.ts +0 -30
  124. package/dist/esm/utilities/messages.js +0 -7
  125. package/dist/esm/utilities/messages.js.map +0 -1
  126. package/dist/esm/utilities/stream-to-response.d.ts +0 -48
  127. package/dist/esm/utilities/stream-to-response.js +0 -62
  128. package/dist/esm/utilities/stream-to-response.js.map +0 -1
  129. package/src/base-adapter.ts +0 -86
  130. package/src/core/chat-common-options.ts +0 -55
  131. package/src/core/embedding.ts +0 -54
  132. package/src/core/summarize.ts +0 -56
  133. package/src/tools/zod-converter.ts +0 -85
  134. package/src/utilities/chat-options.ts +0 -35
  135. package/src/utilities/messages.ts +0 -63
  136. package/src/utilities/stream-to-response.ts +0 -116
  137. /package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.js +0 -0
  138. /package/dist/esm/{stream → activities/chat/stream}/index.d.ts +0 -0
  139. /package/dist/esm/{stream → activities/chat/stream}/json-parser.d.ts +0 -0
  140. /package/dist/esm/{stream → activities/chat/stream}/json-parser.js +0 -0
  141. /package/dist/esm/{stream → activities/chat/stream}/message-updaters.js +0 -0
  142. /package/dist/esm/{stream → activities/chat/stream}/strategies.d.ts +0 -0
  143. /package/dist/esm/{stream → activities/chat/stream}/strategies.js +0 -0
  144. /package/dist/esm/{tools → activities/chat/tools}/tool-definition.js +0 -0
  145. /package/src/{stream → activities/chat/stream}/index.ts +0 -0
  146. /package/src/{stream → activities/chat/stream}/json-parser.ts +0 -0
  147. /package/src/{stream → activities/chat/stream}/strategies.ts +0 -0
package/src/types.ts CHANGED
@@ -1,6 +1,79 @@
1
- import type { CommonOptions } from './core/chat-common-options'
2
1
  import type { z } from 'zod'
3
- import type { ToolCallState, ToolResultState } from './stream/types'
2
+
3
+ /**
4
+ * Tool call states - track the lifecycle of a tool call
5
+ */
6
+ export type ToolCallState =
7
+ | 'awaiting-input' // Received start but no arguments yet
8
+ | 'input-streaming' // Partial arguments received
9
+ | 'input-complete' // All arguments received
10
+ | 'approval-requested' // Waiting for user approval
11
+ | 'approval-responded' // User has approved/denied
12
+
13
+ /**
14
+ * Tool result states - track the lifecycle of a tool result
15
+ */
16
+ export type ToolResultState =
17
+ | 'streaming' // Placeholder for future streamed output
18
+ | 'complete' // Result is complete
19
+ | 'error' // Error occurred
20
+
21
+ /**
22
+ * JSON Schema type for defining tool input/output schemas as raw JSON Schema objects.
23
+ * This allows tools to be defined without Zod when you have JSON Schema definitions available.
24
+ */
25
+ export interface JSONSchema {
26
+ type?: string | Array<string>
27
+ properties?: Record<string, JSONSchema>
28
+ items?: JSONSchema | Array<JSONSchema>
29
+ required?: Array<string>
30
+ enum?: Array<any>
31
+ const?: any
32
+ description?: string
33
+ default?: any
34
+ $ref?: string
35
+ $defs?: Record<string, JSONSchema>
36
+ definitions?: Record<string, JSONSchema>
37
+ allOf?: Array<JSONSchema>
38
+ anyOf?: Array<JSONSchema>
39
+ oneOf?: Array<JSONSchema>
40
+ not?: JSONSchema
41
+ if?: JSONSchema
42
+ then?: JSONSchema
43
+ else?: JSONSchema
44
+ minimum?: number
45
+ maximum?: number
46
+ exclusiveMinimum?: number
47
+ exclusiveMaximum?: number
48
+ minLength?: number
49
+ maxLength?: number
50
+ pattern?: string
51
+ format?: string
52
+ minItems?: number
53
+ maxItems?: number
54
+ uniqueItems?: boolean
55
+ additionalProperties?: boolean | JSONSchema
56
+ additionalItems?: boolean | JSONSchema
57
+ patternProperties?: Record<string, JSONSchema>
58
+ propertyNames?: JSONSchema
59
+ minProperties?: number
60
+ maxProperties?: number
61
+ title?: string
62
+ examples?: Array<any>
63
+ [key: string]: any // Allow additional properties for extensibility
64
+ }
65
+
66
+ /**
67
+ * Union type for schema input - can be either a Zod schema or a JSONSchema object.
68
+ */
69
+ export type SchemaInput = z.ZodType | JSONSchema
70
+
71
+ /**
72
+ * Infer the TypeScript type from a schema.
73
+ * For Zod schemas, uses z.infer to get the proper type.
74
+ * For JSONSchema, returns `any` since we can't infer types from JSON Schema at compile time.
75
+ */
76
+ export type InferSchemaType<T> = T extends z.ZodType ? z.infer<T> : any
4
77
 
5
78
  export interface ToolCall {
6
79
  id: string
@@ -100,11 +173,11 @@ export interface DocumentPart<TMetadata = unknown> {
100
173
  * @template TDocumentMeta - Provider-specific document metadata type
101
174
  */
102
175
  export type ContentPart<
176
+ TTextMeta = unknown,
103
177
  TImageMeta = unknown,
104
178
  TAudioMeta = unknown,
105
179
  TVideoMeta = unknown,
106
180
  TDocumentMeta = unknown,
107
- TTextMeta = unknown,
108
181
  > =
109
182
  | TextPart<TTextMeta>
110
183
  | ImagePart<TImageMeta>
@@ -116,16 +189,17 @@ export type ContentPart<
116
189
  * Helper type to filter ContentPart union to only include specific modalities.
117
190
  * Used to constrain message content based on model capabilities.
118
191
  */
119
- export type ContentPartForModalities<
120
- TModalities extends Modality,
121
- TImageMeta = unknown,
122
- TAudioMeta = unknown,
123
- TVideoMeta = unknown,
124
- TDocumentMeta = unknown,
125
- TTextMeta = unknown,
192
+ export type ContentPartForInputModalitiesTypes<
193
+ TInputModalitiesTypes extends InputModalitiesTypes,
126
194
  > = Extract<
127
- ContentPart<TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>,
128
- { type: TModalities }
195
+ ContentPart<
196
+ TInputModalitiesTypes['messageMetadataByModality']['text'],
197
+ TInputModalitiesTypes['messageMetadataByModality']['image'],
198
+ TInputModalitiesTypes['messageMetadataByModality']['audio'],
199
+ TInputModalitiesTypes['messageMetadataByModality']['video'],
200
+ TInputModalitiesTypes['messageMetadataByModality']['document']
201
+ >,
202
+ { type: TInputModalitiesTypes['inputModalities'][number] }
129
203
  >
130
204
 
131
205
  /**
@@ -140,25 +214,11 @@ export type ModalitiesArrayToUnion<T extends ReadonlyArray<Modality>> =
140
214
  * When modalities is ['text', 'image'], only TextPart and ImagePart are allowed in the array.
141
215
  */
142
216
  export type ConstrainedContent<
143
- TModalities extends ReadonlyArray<Modality>,
144
- TImageMeta = unknown,
145
- TAudioMeta = unknown,
146
- TVideoMeta = unknown,
147
- TDocumentMeta = unknown,
148
- TTextMeta = unknown,
217
+ TInputModalitiesTypes extends InputModalitiesTypes,
149
218
  > =
150
219
  | string
151
220
  | null
152
- | Array<
153
- ContentPartForModalities<
154
- ModalitiesArrayToUnion<TModalities>,
155
- TImageMeta,
156
- TAudioMeta,
157
- TVideoMeta,
158
- TDocumentMeta,
159
- TTextMeta
160
- >
161
- >
221
+ | Array<ContentPartForInputModalitiesTypes<TInputModalitiesTypes>>
162
222
 
163
223
  export interface ModelMessage<
164
224
  TContent extends string | null | Array<ContentPart> =
@@ -227,26 +287,20 @@ export interface UIMessage {
227
287
  parts: Array<MessagePart>
228
288
  createdAt?: Date
229
289
  }
290
+
291
+ export type InputModalitiesTypes = {
292
+ inputModalities: ReadonlyArray<Modality>
293
+ messageMetadataByModality: DefaultMessageMetadataByModality
294
+ }
295
+
230
296
  /**
231
297
  * A ModelMessage with content constrained to only allow content parts
232
298
  * matching the specified input modalities.
233
299
  */
234
300
  export type ConstrainedModelMessage<
235
- TModalities extends ReadonlyArray<Modality>,
236
- TImageMeta = unknown,
237
- TAudioMeta = unknown,
238
- TVideoMeta = unknown,
239
- TDocumentMeta = unknown,
240
- TTextMeta = unknown,
301
+ TInputModalitiesTypes extends InputModalitiesTypes,
241
302
  > = Omit<ModelMessage, 'content'> & {
242
- content: ConstrainedContent<
243
- TModalities,
244
- TImageMeta,
245
- TAudioMeta,
246
- TVideoMeta,
247
- TDocumentMeta,
248
- TTextMeta
249
- >
303
+ content: ConstrainedContent<TInputModalitiesTypes>
250
304
  }
251
305
 
252
306
  /**
@@ -255,14 +309,14 @@ export type ConstrainedModelMessage<
255
309
  * Tools allow the model to interact with external systems, APIs, or perform computations.
256
310
  * The model will decide when to call tools based on the user's request and the tool descriptions.
257
311
  *
258
- * Tools use Zod schemas for runtime validation and type safety.
312
+ * Tools can use either Zod schemas or JSON Schema objects for runtime validation and type safety.
259
313
  *
260
314
  * @see https://platform.openai.com/docs/guides/function-calling
261
315
  * @see https://docs.anthropic.com/claude/docs/tool-use
262
316
  */
263
317
  export interface Tool<
264
- TInput extends z.ZodType = z.ZodType,
265
- TOutput extends z.ZodType = z.ZodType,
318
+ TInput extends SchemaInput = z.ZodType,
319
+ TOutput extends SchemaInput = z.ZodType,
266
320
  TName extends string = string,
267
321
  > {
268
322
  /**
@@ -286,32 +340,47 @@ export interface Tool<
286
340
  description: string
287
341
 
288
342
  /**
289
- * Zod schema describing the tool's input parameters.
343
+ * Schema describing the tool's input parameters.
290
344
  *
345
+ * Can be either a Zod schema or a JSON Schema object.
291
346
  * Defines the structure and types of arguments the tool accepts.
292
347
  * The model will generate arguments matching this schema.
293
- * The schema is converted to JSON Schema for LLM providers.
348
+ * Zod schemas are converted to JSON Schema for LLM providers.
294
349
  *
295
350
  * @see https://zod.dev/
351
+ * @see https://json-schema.org/
296
352
  *
297
353
  * @example
354
+ * // Using Zod schema
298
355
  * import { z } from 'zod';
299
- *
300
356
  * z.object({
301
357
  * location: z.string().describe("City name or coordinates"),
302
358
  * unit: z.enum(["celsius", "fahrenheit"]).optional()
303
359
  * })
360
+ *
361
+ * @example
362
+ * // Using JSON Schema
363
+ * {
364
+ * type: 'object',
365
+ * properties: {
366
+ * location: { type: 'string', description: 'City name or coordinates' },
367
+ * unit: { type: 'string', enum: ['celsius', 'fahrenheit'] }
368
+ * },
369
+ * required: ['location']
370
+ * }
304
371
  */
305
372
  inputSchema?: TInput
306
373
 
307
374
  /**
308
- * Optional Zod schema for validating tool output.
375
+ * Optional schema for validating tool output.
309
376
  *
310
- * If provided, tool results will be validated against this schema before
377
+ * Can be either a Zod schema or a JSON Schema object.
378
+ * If provided with a Zod schema, tool results will be validated against this schema before
311
379
  * being sent back to the model. This catches bugs in tool implementations
312
380
  * and ensures consistent output formatting.
313
381
  *
314
382
  * Note: This is client-side validation only - not sent to LLM providers.
383
+ * Note: JSON Schema output validation is not performed at runtime.
315
384
  *
316
385
  * @example
317
386
  * z.object({
@@ -473,21 +542,71 @@ export type AgentLoopStrategy = (state: AgentLoopState) => boolean
473
542
  /**
474
543
  * Options passed into the SDK and further piped to the AI provider.
475
544
  */
476
- export interface ChatOptions<
477
- TModel extends string = string,
545
+ export interface TextOptions<
478
546
  TProviderOptionsSuperset extends Record<string, any> = Record<string, any>,
479
- TOutput extends ResponseFormat<any> | undefined = undefined,
480
547
  TProviderOptionsForModel = TProviderOptionsSuperset,
481
548
  > {
482
- model: TModel
549
+ model: string
483
550
  messages: Array<ModelMessage>
484
- tools?: Array<Tool>
551
+ tools?: Array<Tool<any, any, any>>
485
552
  systemPrompts?: Array<string>
486
553
  agentLoopStrategy?: AgentLoopStrategy
487
- options?: CommonOptions
488
- providerOptions?: TProviderOptionsForModel
554
+ /**
555
+ * Controls the randomness of the output.
556
+ * Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
557
+ * Range: [0.0, 2.0]
558
+ *
559
+ * Note: Generally recommended to use either temperature or topP, but not both.
560
+ *
561
+ * Provider usage:
562
+ * - OpenAI: `temperature` (number) - in text.top_p field
563
+ * - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
564
+ * - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
565
+ */
566
+ temperature?: number
567
+ /**
568
+ * Nucleus sampling parameter. An alternative to temperature sampling.
569
+ * The model considers the results of tokens with topP probability mass.
570
+ * For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
571
+ *
572
+ * Note: Generally recommended to use either temperature or topP, but not both.
573
+ *
574
+ * Provider usage:
575
+ * - OpenAI: `text.top_p` (number)
576
+ * - Anthropic: `top_p` (number | null)
577
+ * - Gemini: `generationConfig.topP` (number)
578
+ */
579
+ topP?: number
580
+ /**
581
+ * The maximum number of tokens to generate in the response.
582
+ *
583
+ * Provider usage:
584
+ * - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
585
+ * - Anthropic: `max_tokens` (number, required) - range x >= 1
586
+ * - Gemini: `generationConfig.maxOutputTokens` (number)
587
+ */
588
+ maxTokens?: number
589
+ /**
590
+ * Additional metadata to attach to the request.
591
+ * Can be used for tracking, debugging, or passing custom information.
592
+ * Structure and constraints vary by provider.
593
+ *
594
+ * Provider usage:
595
+ * - OpenAI: `metadata` (Record<string, string>) - max 16 key-value pairs, keys max 64 chars, values max 512 chars
596
+ * - Anthropic: `metadata` (Record<string, any>) - includes optional user_id (max 256 chars)
597
+ * - Gemini: Not directly available in TextProviderOptions
598
+ */
599
+ metadata?: Record<string, any>
600
+ modelOptions?: TProviderOptionsForModel
489
601
  request?: Request | RequestInit
490
- output?: TOutput
602
+
603
+ /**
604
+ * Zod schema for structured output.
605
+ * When provided, the adapter should use the provider's native structured output API
606
+ * to ensure the response conforms to this schema.
607
+ * The schema will be converted to JSON Schema format before being sent to the provider.
608
+ */
609
+ outputSchema?: z.ZodType
491
610
  /**
492
611
  * Conversation ID for correlating client and server-side devtools events.
493
612
  * When provided, server-side events will be linked to the client conversation in devtools.
@@ -607,9 +726,9 @@ export type StreamChunk =
607
726
  | ToolInputAvailableStreamChunk
608
727
  | ThinkingStreamChunk
609
728
 
610
- // Simple streaming format for basic chat completions
611
- // Converted to StreamChunk format by convertChatCompletionStream()
612
- export interface ChatCompletionChunk {
729
+ // Simple streaming format for basic text completions
730
+ // Converted to StreamChunk format by convertTextCompletionStream()
731
+ export interface TextCompletionChunk {
613
732
  id: string
614
733
  model: string
615
734
  content: string
@@ -641,245 +760,250 @@ export interface SummarizationResult {
641
760
  }
642
761
  }
643
762
 
644
- export interface EmbeddingOptions {
763
+ // ============================================================================
764
+ // Image Generation Types
765
+ // ============================================================================
766
+
767
+ /**
768
+ * Options for image generation.
769
+ * These are the common options supported across providers.
770
+ */
771
+ export interface ImageGenerationOptions<
772
+ TProviderOptions extends object = object,
773
+ > {
774
+ /** The model to use for image generation */
645
775
  model: string
646
- input: string | Array<string>
647
- dimensions?: number
776
+ /** Text description of the desired image(s) */
777
+ prompt: string
778
+ /** Number of images to generate (default: 1) */
779
+ numberOfImages?: number
780
+ /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
781
+ size?: string
782
+ /** Model-specific options for image generation */
783
+ modelOptions?: TProviderOptions
784
+ }
785
+
786
+ /**
787
+ * A single generated image
788
+ */
789
+ export interface GeneratedImage {
790
+ /** Base64-encoded image data */
791
+ b64Json?: string
792
+ /** URL to the generated image (may be temporary) */
793
+ url?: string
794
+ /** Revised prompt used by the model (if applicable) */
795
+ revisedPrompt?: string
648
796
  }
649
797
 
650
- export interface EmbeddingResult {
798
+ /**
799
+ * Result of image generation
800
+ */
801
+ export interface ImageGenerationResult {
802
+ /** Unique identifier for the generation */
651
803
  id: string
804
+ /** Model used for generation */
652
805
  model: string
653
- embeddings: Array<Array<number>>
654
- usage: {
655
- promptTokens: number
656
- totalTokens: number
806
+ /** Array of generated images */
807
+ images: Array<GeneratedImage>
808
+ /** Token usage information (if available) */
809
+ usage?: {
810
+ inputTokens?: number
811
+ outputTokens?: number
812
+ totalTokens?: number
657
813
  }
658
814
  }
659
815
 
816
+ // ============================================================================
817
+ // Video Generation Types (Experimental)
818
+ // ============================================================================
819
+
660
820
  /**
661
- * Default metadata type for adapters that don't define custom metadata.
662
- * Uses unknown for all modalities.
821
+ * Options for video generation.
822
+ * These are the common options supported across providers.
823
+ *
824
+ * @experimental Video generation is an experimental feature and may change.
663
825
  */
664
- export interface DefaultMessageMetadataByModality {
665
- text: unknown
666
- image: unknown
667
- audio: unknown
668
- video: unknown
669
- document: unknown
826
+ export interface VideoGenerationOptions<
827
+ TProviderOptions extends object = object,
828
+ > {
829
+ /** The model to use for video generation */
830
+ model: string
831
+ /** Text description of the desired video */
832
+ prompt: string
833
+ /** Video size in WIDTHxHEIGHT format (e.g., "1280x720") */
834
+ size?: string
835
+ /** Video duration in seconds */
836
+ duration?: number
837
+ /** Model-specific options for video generation */
838
+ modelOptions?: TProviderOptions
670
839
  }
671
840
 
672
841
  /**
673
- * AI adapter interface with support for endpoint-specific models and provider options.
842
+ * Result of creating a video generation job.
674
843
  *
675
- * Generic parameters:
676
- * - TChatModels: Models that support chat/text completion
677
- * - TEmbeddingModels: Models that support embeddings
678
- * - TChatProviderOptions: Provider-specific options for chat endpoint
679
- * - TEmbeddingProviderOptions: Provider-specific options for embedding endpoint
680
- * - TModelProviderOptionsByName: Map from model name to its specific provider options
681
- * - TModelInputModalitiesByName: Map from model name to its supported input modalities
682
- * - TMessageMetadataByModality: Map from modality type to adapter-specific metadata types
683
- */
684
- export interface AIAdapter<
685
- TChatModels extends ReadonlyArray<string> = ReadonlyArray<string>,
686
- TEmbeddingModels extends ReadonlyArray<string> = ReadonlyArray<string>,
687
- TChatProviderOptions extends Record<string, any> = Record<string, any>,
688
- TEmbeddingProviderOptions extends Record<string, any> = Record<string, any>,
689
- TModelProviderOptionsByName extends Record<string, any> = Record<string, any>,
690
- TModelInputModalitiesByName extends Record<
691
- string,
692
- ReadonlyArray<Modality>
693
- > = Record<string, ReadonlyArray<Modality>>,
694
- TMessageMetadataByModality extends {
695
- text: unknown
696
- image: unknown
697
- audio: unknown
698
- video: unknown
699
- document: unknown
700
- } = DefaultMessageMetadataByModality,
701
- > {
702
- name: string
703
- /** Models that support chat/text completion */
704
- models: TChatModels
844
+ * @experimental Video generation is an experimental feature and may change.
845
+ */
846
+ export interface VideoJobResult {
847
+ /** Unique job identifier for polling status */
848
+ jobId: string
849
+ /** Model used for generation */
850
+ model: string
851
+ }
705
852
 
706
- /** Models that support embeddings */
707
- embeddingModels?: TEmbeddingModels
853
+ /**
854
+ * Status of a video generation job.
855
+ *
856
+ * @experimental Video generation is an experimental feature and may change.
857
+ */
858
+ export interface VideoStatusResult {
859
+ /** Job identifier */
860
+ jobId: string
861
+ /** Current status of the job */
862
+ status: 'pending' | 'processing' | 'completed' | 'failed'
863
+ /** Progress percentage (0-100), if available */
864
+ progress?: number
865
+ /** Error message if status is 'failed' */
866
+ error?: string
867
+ }
708
868
 
709
- // Type-only properties for provider options inference
710
- _providerOptions?: TChatProviderOptions // Alias for _chatProviderOptions
711
- _chatProviderOptions?: TChatProviderOptions
712
- _embeddingProviderOptions?: TEmbeddingProviderOptions
713
- /**
714
- * Type-only map from model name to its specific provider options.
715
- * Used by the core AI types to narrow providerOptions based on the selected model.
716
- * Must be provided by all adapters.
717
- */
718
- _modelProviderOptionsByName: TModelProviderOptionsByName
719
- /**
720
- * Type-only map from model name to its supported input modalities.
721
- * Used by the core AI types to narrow ContentPart types based on the selected model.
722
- * Must be provided by all adapters.
723
- */
724
- _modelInputModalitiesByName?: TModelInputModalitiesByName
725
- /**
726
- * Type-only map from modality type to adapter-specific metadata types.
727
- * Used to provide type-safe autocomplete for metadata on content parts.
728
- */
729
- _messageMetadataByModality?: TMessageMetadataByModality
869
+ /**
870
+ * Result containing the URL to a generated video.
871
+ *
872
+ * @experimental Video generation is an experimental feature and may change.
873
+ */
874
+ export interface VideoUrlResult {
875
+ /** Job identifier */
876
+ jobId: string
877
+ /** URL to the generated video */
878
+ url: string
879
+ /** When the URL expires, if applicable */
880
+ expiresAt?: Date
881
+ }
882
+
883
+ // ============================================================================
884
+ // Text-to-Speech (TTS) Types
885
+ // ============================================================================
886
+
887
+ /**
888
+ * Options for text-to-speech generation.
889
+ * These are the common options supported across providers.
890
+ */
891
+ export interface TTSOptions<TProviderOptions extends object = object> {
892
+ /** The model to use for TTS generation */
893
+ model: string
894
+ /** The text to convert to speech */
895
+ text: string
896
+ /** The voice to use for generation */
897
+ voice?: string
898
+ /** The output audio format */
899
+ format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'
900
+ /** The speed of the generated audio (0.25 to 4.0) */
901
+ speed?: number
902
+ /** Model-specific options for TTS generation */
903
+ modelOptions?: TProviderOptions
904
+ }
730
905
 
731
- // Structured streaming with JSON chunks (supports tool calls and rich content)
732
- chatStream: (
733
- options: ChatOptions<string, TChatProviderOptions>,
734
- ) => AsyncIterable<StreamChunk>
906
+ /**
907
+ * Result of text-to-speech generation.
908
+ */
909
+ export interface TTSResult {
910
+ /** Unique identifier for the generation */
911
+ id: string
912
+ /** Model used for generation */
913
+ model: string
914
+ /** Base64-encoded audio data */
915
+ audio: string
916
+ /** Audio format of the generated audio */
917
+ format: string
918
+ /** Duration of the audio in seconds, if available */
919
+ duration?: number
920
+ /** Content type of the audio (e.g., 'audio/mp3') */
921
+ contentType?: string
922
+ }
735
923
 
736
- // Summarization
737
- summarize: (options: SummarizationOptions) => Promise<SummarizationResult>
924
+ // ============================================================================
925
+ // Transcription (Speech-to-Text) Types
926
+ // ============================================================================
738
927
 
739
- // Embeddings
740
- createEmbeddings: (options: EmbeddingOptions) => Promise<EmbeddingResult>
928
+ /**
929
+ * Options for audio transcription.
930
+ * These are the common options supported across providers.
931
+ */
932
+ export interface TranscriptionOptions<
933
+ TProviderOptions extends object = object,
934
+ > {
935
+ /** The model to use for transcription */
936
+ model: string
937
+ /** The audio data to transcribe - can be base64 string, File, Blob, or Buffer */
938
+ audio: string | File | Blob | ArrayBuffer
939
+ /** The language of the audio in ISO-639-1 format (e.g., 'en') */
940
+ language?: string
941
+ /** An optional prompt to guide the transcription */
942
+ prompt?: string
943
+ /** The format of the transcription output */
944
+ responseFormat?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt'
945
+ /** Model-specific options for transcription */
946
+ modelOptions?: TProviderOptions
741
947
  }
742
948
 
743
- export interface AIAdapterConfig {
744
- apiKey?: string
745
- baseUrl?: string
746
- timeout?: number
747
- maxRetries?: number
748
- headers?: Record<string, string>
949
+ /**
950
+ * A single segment of transcribed audio with timing information.
951
+ */
952
+ export interface TranscriptionSegment {
953
+ /** Unique identifier for the segment */
954
+ id: number
955
+ /** Start time of the segment in seconds */
956
+ start: number
957
+ /** End time of the segment in seconds */
958
+ end: number
959
+ /** Transcribed text for this segment */
960
+ text: string
961
+ /** Confidence score (0-1), if available */
962
+ confidence?: number
963
+ /** Speaker identifier, if diarization is enabled */
964
+ speaker?: string
749
965
  }
750
966
 
751
- export type ChatStreamOptionsUnion<
752
- TAdapter extends AIAdapter<any, any, any, any, any, any, any>,
753
- > =
754
- TAdapter extends AIAdapter<
755
- infer Models,
756
- any,
757
- any,
758
- any,
759
- infer ModelProviderOptions,
760
- infer ModelInputModalities,
761
- infer MessageMetadata
762
- >
763
- ? Models[number] extends infer TModel
764
- ? TModel extends string
765
- ? Omit<
766
- ChatOptions,
767
- 'model' | 'providerOptions' | 'responseFormat' | 'messages'
768
- > & {
769
- adapter: TAdapter
770
- model: TModel
771
- providerOptions?: TModel extends keyof ModelProviderOptions
772
- ? ModelProviderOptions[TModel]
773
- : never
774
- /**
775
- * Messages array with content constrained to the model's supported input modalities.
776
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
777
- * Metadata types are also constrained based on the adapter's metadata type definitions.
778
- */
779
- messages: TModel extends keyof ModelInputModalities
780
- ? ModelInputModalities[TModel] extends ReadonlyArray<Modality>
781
- ? MessageMetadata extends {
782
- text: infer TTextMeta
783
- image: infer TImageMeta
784
- audio: infer TAudioMeta
785
- video: infer TVideoMeta
786
- document: infer TDocumentMeta
787
- }
788
- ? Array<
789
- ConstrainedModelMessage<
790
- ModelInputModalities[TModel],
791
- TImageMeta,
792
- TAudioMeta,
793
- TVideoMeta,
794
- TDocumentMeta,
795
- TTextMeta
796
- >
797
- >
798
- : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>>
799
- : Array<ModelMessage>
800
- : Array<ModelMessage>
801
- }
802
- : never
803
- : never
804
- : never
805
-
806
- /**
807
- * Chat options constrained by a specific model's capabilities.
808
- * Unlike ChatStreamOptionsUnion which creates a union over all models,
809
- * this type takes a specific model and constrains messages accordingly.
810
- */
811
- export type ChatStreamOptionsForModel<
812
- TAdapter extends AIAdapter<any, any, any, any, any, any, any>,
813
- TModel extends string,
814
- > =
815
- TAdapter extends AIAdapter<
816
- any,
817
- any,
818
- any,
819
- any,
820
- infer ModelProviderOptions,
821
- infer ModelInputModalities,
822
- infer MessageMetadata
823
- >
824
- ? Omit<
825
- ChatOptions,
826
- 'model' | 'providerOptions' | 'responseFormat' | 'messages'
827
- > & {
828
- adapter: TAdapter
829
- model: TModel
830
- providerOptions?: TModel extends keyof ModelProviderOptions
831
- ? ModelProviderOptions[TModel]
832
- : never
833
- /**
834
- * Messages array with content constrained to the model's supported input modalities.
835
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
836
- * Metadata types are also constrained based on the adapter's metadata type definitions.
837
- */
838
- messages: TModel extends keyof ModelInputModalities
839
- ? ModelInputModalities[TModel] extends ReadonlyArray<Modality>
840
- ? MessageMetadata extends {
841
- text: infer TTextMeta
842
- image: infer TImageMeta
843
- audio: infer TAudioMeta
844
- video: infer TVideoMeta
845
- document: infer TDocumentMeta
846
- }
847
- ? Array<
848
- ConstrainedModelMessage<
849
- ModelInputModalities[TModel],
850
- TImageMeta,
851
- TAudioMeta,
852
- TVideoMeta,
853
- TDocumentMeta,
854
- TTextMeta
855
- >
856
- >
857
- : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>>
858
- : Array<ModelMessage>
859
- : Array<ModelMessage>
860
- }
861
- : never
862
-
863
- // Extract types from adapter (updated to 6 generics)
864
- export type ExtractModelsFromAdapter<T> =
865
- T extends AIAdapter<infer M, any, any, any, any, any> ? M[number] : never
866
-
867
- /**
868
- * Extract the supported input modalities for a specific model from an adapter.
869
- */
870
- export type ExtractModalitiesForModel<
871
- TAdapter extends AIAdapter<any, any, any, any, any, any>,
872
- TModel extends string,
873
- > =
874
- TAdapter extends AIAdapter<
875
- any,
876
- any,
877
- any,
878
- any,
879
- any,
880
- infer ModelInputModalities
881
- >
882
- ? TModel extends keyof ModelInputModalities
883
- ? ModelInputModalities[TModel]
884
- : ReadonlyArray<Modality>
885
- : ReadonlyArray<Modality>
967
+ /**
968
+ * A single word with timing information.
969
+ */
970
+ export interface TranscriptionWord {
971
+ /** The transcribed word */
972
+ word: string
973
+ /** Start time in seconds */
974
+ start: number
975
+ /** End time in seconds */
976
+ end: number
977
+ }
978
+
979
+ /**
980
+ * Result of audio transcription.
981
+ */
982
+ export interface TranscriptionResult {
983
+ /** Unique identifier for the transcription */
984
+ id: string
985
+ /** Model used for transcription */
986
+ model: string
987
+ /** The full transcribed text */
988
+ text: string
989
+ /** Language detected or specified */
990
+ language?: string
991
+ /** Duration of the audio in seconds */
992
+ duration?: number
993
+ /** Detailed segments with timing, if available */
994
+ segments?: Array<TranscriptionSegment>
995
+ /** Word-level timestamps, if available */
996
+ words?: Array<TranscriptionWord>
997
+ }
998
+
999
+ /**
1000
+ * Default metadata type for adapters that don't define custom metadata.
1001
+ * Uses unknown for all modalities.
1002
+ */
1003
+ export interface DefaultMessageMetadataByModality {
1004
+ text: unknown
1005
+ image: unknown
1006
+ audio: unknown
1007
+ video: unknown
1008
+ document: unknown
1009
+ }