@tanstack/ai 0.0.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/README.md +26 -0
  2. package/dist/esm/activities/chat/adapter.d.ts +100 -0
  3. package/dist/esm/activities/chat/adapter.js +14 -0
  4. package/dist/esm/activities/chat/adapter.js.map +1 -0
  5. package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.d.ts +4 -4
  6. package/dist/esm/activities/chat/agent-loop-strategies.js.map +1 -0
  7. package/dist/esm/activities/chat/index.d.ts +166 -0
  8. package/dist/esm/{core/chat.js → activities/chat/index.js} +131 -33
  9. package/dist/esm/activities/chat/index.js.map +1 -0
  10. package/dist/esm/{message-converters.d.ts → activities/chat/messages.d.ts} +1 -1
  11. package/dist/esm/{message-converters.js → activities/chat/messages.js} +7 -7
  12. package/dist/esm/activities/chat/messages.js.map +1 -0
  13. package/dist/esm/activities/chat/stream/json-parser.js.map +1 -0
  14. package/dist/esm/{stream → activities/chat/stream}/message-updaters.d.ts +1 -1
  15. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -0
  16. package/dist/esm/{stream → activities/chat/stream}/processor.d.ts +1 -1
  17. package/dist/esm/{stream → activities/chat/stream}/processor.js +1 -1
  18. package/dist/esm/activities/chat/stream/processor.js.map +1 -0
  19. package/dist/esm/activities/chat/stream/strategies.js.map +1 -0
  20. package/dist/esm/{stream → activities/chat/stream}/types.d.ts +2 -9
  21. package/dist/esm/activities/chat/tools/schema-converter.d.ts +116 -0
  22. package/dist/esm/activities/chat/tools/schema-converter.js +115 -0
  23. package/dist/esm/activities/chat/tools/schema-converter.js.map +1 -0
  24. package/dist/esm/{tools → activities/chat/tools}/tool-calls.d.ts +1 -1
  25. package/dist/esm/{tools → activities/chat/tools}/tool-calls.js +23 -30
  26. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -0
  27. package/dist/esm/{tools → activities/chat/tools}/tool-definition.d.ts +22 -18
  28. package/dist/esm/activities/chat/tools/tool-definition.js.map +1 -0
  29. package/dist/esm/activities/generateImage/adapter.d.ts +68 -0
  30. package/dist/esm/activities/generateImage/adapter.js +14 -0
  31. package/dist/esm/activities/generateImage/adapter.js.map +1 -0
  32. package/dist/esm/activities/generateImage/index.d.ts +89 -0
  33. package/dist/esm/activities/generateImage/index.js +15 -0
  34. package/dist/esm/activities/generateImage/index.js.map +1 -0
  35. package/dist/esm/activities/generateSpeech/adapter.d.ts +62 -0
  36. package/dist/esm/activities/generateSpeech/adapter.js +14 -0
  37. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -0
  38. package/dist/esm/activities/generateSpeech/index.d.ts +69 -0
  39. package/dist/esm/activities/generateSpeech/index.js +15 -0
  40. package/dist/esm/activities/generateSpeech/index.js.map +1 -0
  41. package/dist/esm/activities/generateTranscription/adapter.d.ts +62 -0
  42. package/dist/esm/activities/generateTranscription/adapter.js +14 -0
  43. package/dist/esm/activities/generateTranscription/adapter.js.map +1 -0
  44. package/dist/esm/activities/generateTranscription/index.d.ts +71 -0
  45. package/dist/esm/activities/generateTranscription/index.js +15 -0
  46. package/dist/esm/activities/generateTranscription/index.js.map +1 -0
  47. package/dist/esm/activities/generateVideo/adapter.d.ts +80 -0
  48. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  49. package/dist/esm/activities/generateVideo/adapter.js.map +1 -0
  50. package/dist/esm/activities/generateVideo/index.d.ts +136 -0
  51. package/dist/esm/activities/generateVideo/index.js +47 -0
  52. package/dist/esm/activities/generateVideo/index.js.map +1 -0
  53. package/dist/esm/activities/index.d.ts +22 -0
  54. package/dist/esm/activities/index.js +34 -0
  55. package/dist/esm/activities/index.js.map +1 -0
  56. package/dist/esm/activities/summarize/adapter.d.ts +74 -0
  57. package/dist/esm/activities/summarize/adapter.js +14 -0
  58. package/dist/esm/activities/summarize/adapter.js.map +1 -0
  59. package/dist/esm/activities/summarize/index.d.ts +100 -0
  60. package/dist/esm/activities/summarize/index.js +90 -0
  61. package/dist/esm/activities/summarize/index.js.map +1 -0
  62. package/dist/esm/event-client.d.ts +4 -18
  63. package/dist/esm/event-client.js.map +1 -1
  64. package/dist/esm/index.d.ts +16 -14
  65. package/dist/esm/index.js +29 -20
  66. package/dist/esm/stream-to-response.d.ts +95 -0
  67. package/dist/esm/stream-to-response.js +118 -0
  68. package/dist/esm/stream-to-response.js.map +1 -0
  69. package/dist/esm/types.d.ts +370 -133
  70. package/package.json +7 -6
  71. package/src/activities/chat/adapter.ts +150 -0
  72. package/src/{utilities → activities/chat}/agent-loop-strategies.ts +4 -4
  73. package/src/{core/chat.ts → activities/chat/index.ts} +435 -79
  74. package/src/{message-converters.ts → activities/chat/messages.ts} +10 -13
  75. package/src/{stream → activities/chat/stream}/message-updaters.ts +1 -1
  76. package/src/{stream → activities/chat/stream}/processor.ts +2 -5
  77. package/src/{stream → activities/chat/stream}/types.ts +8 -18
  78. package/src/activities/chat/tools/schema-converter.ts +332 -0
  79. package/src/{tools → activities/chat/tools}/tool-calls.ts +63 -44
  80. package/src/{tools → activities/chat/tools}/tool-definition.ts +51 -38
  81. package/src/activities/generateImage/adapter.ts +104 -0
  82. package/src/activities/generateImage/index.ts +162 -0
  83. package/src/activities/generateSpeech/adapter.ts +87 -0
  84. package/src/activities/generateSpeech/index.ts +122 -0
  85. package/src/activities/generateTranscription/adapter.ts +89 -0
  86. package/src/activities/generateTranscription/index.ts +132 -0
  87. package/src/activities/generateVideo/adapter.ts +116 -0
  88. package/src/activities/generateVideo/index.ts +261 -0
  89. package/src/activities/index.ts +164 -0
  90. package/src/activities/summarize/adapter.ts +107 -0
  91. package/src/activities/summarize/index.ts +287 -0
  92. package/src/event-client.ts +5 -21
  93. package/src/index.ts +60 -15
  94. package/src/stream-to-response.ts +237 -0
  95. package/src/types.ts +429 -284
  96. package/dist/esm/base-adapter.d.ts +0 -36
  97. package/dist/esm/base-adapter.js +0 -12
  98. package/dist/esm/base-adapter.js.map +0 -1
  99. package/dist/esm/core/chat-common-options.d.ts +0 -52
  100. package/dist/esm/core/chat.d.ts +0 -30
  101. package/dist/esm/core/chat.js.map +0 -1
  102. package/dist/esm/core/embedding.d.ts +0 -8
  103. package/dist/esm/core/embedding.js +0 -33
  104. package/dist/esm/core/embedding.js.map +0 -1
  105. package/dist/esm/core/summarize.d.ts +0 -9
  106. package/dist/esm/core/summarize.js +0 -36
  107. package/dist/esm/core/summarize.js.map +0 -1
  108. package/dist/esm/message-converters.js.map +0 -1
  109. package/dist/esm/stream/json-parser.js.map +0 -1
  110. package/dist/esm/stream/message-updaters.js.map +0 -1
  111. package/dist/esm/stream/processor.js.map +0 -1
  112. package/dist/esm/stream/strategies.js.map +0 -1
  113. package/dist/esm/tools/tool-calls.js.map +0 -1
  114. package/dist/esm/tools/tool-definition.js.map +0 -1
  115. package/dist/esm/tools/zod-converter.d.ts +0 -30
  116. package/dist/esm/tools/zod-converter.js +0 -36
  117. package/dist/esm/tools/zod-converter.js.map +0 -1
  118. package/dist/esm/utilities/agent-loop-strategies.js.map +0 -1
  119. package/dist/esm/utilities/chat-options.d.ts +0 -6
  120. package/dist/esm/utilities/chat-options.js +0 -7
  121. package/dist/esm/utilities/chat-options.js.map +0 -1
  122. package/dist/esm/utilities/messages.d.ts +0 -30
  123. package/dist/esm/utilities/messages.js +0 -7
  124. package/dist/esm/utilities/messages.js.map +0 -1
  125. package/dist/esm/utilities/stream-to-response.d.ts +0 -48
  126. package/dist/esm/utilities/stream-to-response.js +0 -62
  127. package/dist/esm/utilities/stream-to-response.js.map +0 -1
  128. package/src/base-adapter.ts +0 -86
  129. package/src/core/chat-common-options.ts +0 -55
  130. package/src/core/embedding.ts +0 -54
  131. package/src/core/summarize.ts +0 -56
  132. package/src/tools/zod-converter.ts +0 -85
  133. package/src/utilities/chat-options.ts +0 -35
  134. package/src/utilities/messages.ts +0 -63
  135. package/src/utilities/stream-to-response.ts +0 -116
  136. /package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.js +0 -0
  137. /package/dist/esm/{stream → activities/chat/stream}/index.d.ts +0 -0
  138. /package/dist/esm/{stream → activities/chat/stream}/json-parser.d.ts +0 -0
  139. /package/dist/esm/{stream → activities/chat/stream}/json-parser.js +0 -0
  140. /package/dist/esm/{stream → activities/chat/stream}/message-updaters.js +0 -0
  141. /package/dist/esm/{stream → activities/chat/stream}/strategies.d.ts +0 -0
  142. /package/dist/esm/{stream → activities/chat/stream}/strategies.js +0 -0
  143. /package/dist/esm/{tools → activities/chat/tools}/tool-definition.js +0 -0
  144. /package/src/{stream → activities/chat/stream}/index.ts +0 -0
  145. /package/src/{stream → activities/chat/stream}/json-parser.ts +0 -0
  146. /package/src/{stream → activities/chat/stream}/strategies.ts +0 -0
package/src/types.ts CHANGED
@@ -1,6 +1,88 @@
1
- import type { CommonOptions } from './core/chat-common-options'
2
- import type { z } from 'zod'
3
- import type { ToolCallState, ToolResultState } from './stream/types'
1
+ import type { StandardJSONSchemaV1 } from '@standard-schema/spec'
2
+
3
+ /**
4
+ * Tool call states - track the lifecycle of a tool call
5
+ */
6
+ export type ToolCallState =
7
+ | 'awaiting-input' // Received start but no arguments yet
8
+ | 'input-streaming' // Partial arguments received
9
+ | 'input-complete' // All arguments received
10
+ | 'approval-requested' // Waiting for user approval
11
+ | 'approval-responded' // User has approved/denied
12
+
13
+ /**
14
+ * Tool result states - track the lifecycle of a tool result
15
+ */
16
+ export type ToolResultState =
17
+ | 'streaming' // Placeholder for future streamed output
18
+ | 'complete' // Result is complete
19
+ | 'error' // Error occurred
20
+
21
+ /**
22
+ * JSON Schema type for defining tool input/output schemas as raw JSON Schema objects.
23
+ * This allows tools to be defined without schema libraries when you have JSON Schema definitions available.
24
+ */
25
+ export interface JSONSchema {
26
+ type?: string | Array<string>
27
+ properties?: Record<string, JSONSchema>
28
+ items?: JSONSchema | Array<JSONSchema>
29
+ required?: Array<string>
30
+ enum?: Array<unknown>
31
+ const?: unknown
32
+ description?: string
33
+ default?: unknown
34
+ $ref?: string
35
+ $defs?: Record<string, JSONSchema>
36
+ definitions?: Record<string, JSONSchema>
37
+ allOf?: Array<JSONSchema>
38
+ anyOf?: Array<JSONSchema>
39
+ oneOf?: Array<JSONSchema>
40
+ not?: JSONSchema
41
+ if?: JSONSchema
42
+ then?: JSONSchema
43
+ else?: JSONSchema
44
+ minimum?: number
45
+ maximum?: number
46
+ exclusiveMinimum?: number
47
+ exclusiveMaximum?: number
48
+ minLength?: number
49
+ maxLength?: number
50
+ pattern?: string
51
+ format?: string
52
+ minItems?: number
53
+ maxItems?: number
54
+ uniqueItems?: boolean
55
+ additionalProperties?: boolean | JSONSchema
56
+ additionalItems?: boolean | JSONSchema
57
+ patternProperties?: Record<string, JSONSchema>
58
+ propertyNames?: JSONSchema
59
+ minProperties?: number
60
+ maxProperties?: number
61
+ title?: string
62
+ examples?: Array<unknown>
63
+ [key: string]: unknown // Allow additional properties for extensibility
64
+ }
65
+
66
+ /**
67
+ * Union type for schema input - can be any Standard JSON Schema compliant schema or a plain JSONSchema object.
68
+ *
69
+ * Standard JSON Schema compliant libraries include:
70
+ * - Zod v4.2+ (natively supports StandardJSONSchemaV1)
71
+ * - ArkType v2.1.28+ (natively supports StandardJSONSchemaV1)
72
+ * - Valibot v1.2+ (via `toStandardJsonSchema()` from `@valibot/to-json-schema`)
73
+ *
74
+ * @see https://standardschema.dev/json-schema
75
+ */
76
+
77
+ export type SchemaInput = StandardJSONSchemaV1<any, any> | JSONSchema
78
+
79
+ /**
80
+ * Infer the TypeScript type from a schema.
81
+ * For Standard JSON Schema compliant schemas, extracts the input type.
82
+ * For plain JSONSchema, returns `any` since we can't infer types from JSON Schema at compile time.
83
+ */
84
+ export type InferSchemaType<T> =
85
+ T extends StandardJSONSchemaV1<infer TInput, unknown> ? TInput : unknown
4
86
 
5
87
  export interface ToolCall {
6
88
  id: string
@@ -100,11 +182,11 @@ export interface DocumentPart<TMetadata = unknown> {
100
182
  * @template TDocumentMeta - Provider-specific document metadata type
101
183
  */
102
184
  export type ContentPart<
185
+ TTextMeta = unknown,
103
186
  TImageMeta = unknown,
104
187
  TAudioMeta = unknown,
105
188
  TVideoMeta = unknown,
106
189
  TDocumentMeta = unknown,
107
- TTextMeta = unknown,
108
190
  > =
109
191
  | TextPart<TTextMeta>
110
192
  | ImagePart<TImageMeta>
@@ -116,16 +198,17 @@ export type ContentPart<
116
198
  * Helper type to filter ContentPart union to only include specific modalities.
117
199
  * Used to constrain message content based on model capabilities.
118
200
  */
119
- export type ContentPartForModalities<
120
- TModalities extends Modality,
121
- TImageMeta = unknown,
122
- TAudioMeta = unknown,
123
- TVideoMeta = unknown,
124
- TDocumentMeta = unknown,
125
- TTextMeta = unknown,
201
+ export type ContentPartForInputModalitiesTypes<
202
+ TInputModalitiesTypes extends InputModalitiesTypes,
126
203
  > = Extract<
127
- ContentPart<TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>,
128
- { type: TModalities }
204
+ ContentPart<
205
+ TInputModalitiesTypes['messageMetadataByModality']['text'],
206
+ TInputModalitiesTypes['messageMetadataByModality']['image'],
207
+ TInputModalitiesTypes['messageMetadataByModality']['audio'],
208
+ TInputModalitiesTypes['messageMetadataByModality']['video'],
209
+ TInputModalitiesTypes['messageMetadataByModality']['document']
210
+ >,
211
+ { type: TInputModalitiesTypes['inputModalities'][number] }
129
212
  >
130
213
 
131
214
  /**
@@ -140,25 +223,11 @@ export type ModalitiesArrayToUnion<T extends ReadonlyArray<Modality>> =
140
223
  * When modalities is ['text', 'image'], only TextPart and ImagePart are allowed in the array.
141
224
  */
142
225
  export type ConstrainedContent<
143
- TModalities extends ReadonlyArray<Modality>,
144
- TImageMeta = unknown,
145
- TAudioMeta = unknown,
146
- TVideoMeta = unknown,
147
- TDocumentMeta = unknown,
148
- TTextMeta = unknown,
226
+ TInputModalitiesTypes extends InputModalitiesTypes,
149
227
  > =
150
228
  | string
151
229
  | null
152
- | Array<
153
- ContentPartForModalities<
154
- ModalitiesArrayToUnion<TModalities>,
155
- TImageMeta,
156
- TAudioMeta,
157
- TVideoMeta,
158
- TDocumentMeta,
159
- TTextMeta
160
- >
161
- >
230
+ | Array<ContentPartForInputModalitiesTypes<TInputModalitiesTypes>>
162
231
 
163
232
  export interface ModelMessage<
164
233
  TContent extends string | null | Array<ContentPart> =
@@ -227,26 +296,20 @@ export interface UIMessage {
227
296
  parts: Array<MessagePart>
228
297
  createdAt?: Date
229
298
  }
299
+
300
+ export type InputModalitiesTypes = {
301
+ inputModalities: ReadonlyArray<Modality>
302
+ messageMetadataByModality: DefaultMessageMetadataByModality
303
+ }
304
+
230
305
  /**
231
306
  * A ModelMessage with content constrained to only allow content parts
232
307
  * matching the specified input modalities.
233
308
  */
234
309
  export type ConstrainedModelMessage<
235
- TModalities extends ReadonlyArray<Modality>,
236
- TImageMeta = unknown,
237
- TAudioMeta = unknown,
238
- TVideoMeta = unknown,
239
- TDocumentMeta = unknown,
240
- TTextMeta = unknown,
310
+ TInputModalitiesTypes extends InputModalitiesTypes,
241
311
  > = Omit<ModelMessage, 'content'> & {
242
- content: ConstrainedContent<
243
- TModalities,
244
- TImageMeta,
245
- TAudioMeta,
246
- TVideoMeta,
247
- TDocumentMeta,
248
- TTextMeta
249
- >
312
+ content: ConstrainedContent<TInputModalitiesTypes>
250
313
  }
251
314
 
252
315
  /**
@@ -255,14 +318,16 @@ export type ConstrainedModelMessage<
255
318
  * Tools allow the model to interact with external systems, APIs, or perform computations.
256
319
  * The model will decide when to call tools based on the user's request and the tool descriptions.
257
320
  *
258
- * Tools use Zod schemas for runtime validation and type safety.
321
+ * Tools can use any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.)
322
+ * or plain JSON Schema objects for runtime validation and type safety.
259
323
  *
260
324
  * @see https://platform.openai.com/docs/guides/function-calling
261
325
  * @see https://docs.anthropic.com/claude/docs/tool-use
326
+ * @see https://standardschema.dev/json-schema
262
327
  */
263
328
  export interface Tool<
264
- TInput extends z.ZodType = z.ZodType,
265
- TOutput extends z.ZodType = z.ZodType,
329
+ TInput extends SchemaInput = SchemaInput,
330
+ TOutput extends SchemaInput = SchemaInput,
266
331
  TName extends string = string,
267
332
  > {
268
333
  /**
@@ -286,34 +351,58 @@ export interface Tool<
286
351
  description: string
287
352
 
288
353
  /**
289
- * Zod schema describing the tool's input parameters.
354
+ * Schema describing the tool's input parameters.
290
355
  *
356
+ * Can be any Standard JSON Schema compliant schema (Zod, ArkType, Valibot, etc.) or a plain JSON Schema object.
291
357
  * Defines the structure and types of arguments the tool accepts.
292
358
  * The model will generate arguments matching this schema.
293
- * The schema is converted to JSON Schema for LLM providers.
359
+ * Standard JSON Schema compliant schemas are converted to JSON Schema for LLM providers.
294
360
  *
295
- * @see https://zod.dev/
361
+ * @see https://standardschema.dev/json-schema
362
+ * @see https://json-schema.org/
296
363
  *
297
364
  * @example
365
+ * // Using Zod v4+ schema (natively supports Standard JSON Schema)
298
366
  * import { z } from 'zod';
299
- *
300
367
  * z.object({
301
368
  * location: z.string().describe("City name or coordinates"),
302
369
  * unit: z.enum(["celsius", "fahrenheit"]).optional()
303
370
  * })
371
+ *
372
+ * @example
373
+ * // Using ArkType (natively supports Standard JSON Schema)
374
+ * import { type } from 'arktype';
375
+ * type({
376
+ * location: 'string',
377
+ * unit: "'celsius' | 'fahrenheit'"
378
+ * })
379
+ *
380
+ * @example
381
+ * // Using plain JSON Schema
382
+ * {
383
+ * type: 'object',
384
+ * properties: {
385
+ * location: { type: 'string', description: 'City name or coordinates' },
386
+ * unit: { type: 'string', enum: ['celsius', 'fahrenheit'] }
387
+ * },
388
+ * required: ['location']
389
+ * }
304
390
  */
305
391
  inputSchema?: TInput
306
392
 
307
393
  /**
308
- * Optional Zod schema for validating tool output.
394
+ * Optional schema for validating tool output.
309
395
  *
310
- * If provided, tool results will be validated against this schema before
311
- * being sent back to the model. This catches bugs in tool implementations
312
- * and ensures consistent output formatting.
396
+ * Can be any Standard JSON Schema compliant schema or a plain JSON Schema object.
397
+ * If provided with a Standard Schema compliant schema, tool results will be validated
398
+ * against this schema before being sent back to the model. This catches bugs in tool
399
+ * implementations and ensures consistent output formatting.
313
400
  *
314
401
  * Note: This is client-side validation only - not sent to LLM providers.
402
+ * Note: Plain JSON Schema output validation is not performed at runtime.
315
403
  *
316
404
  * @example
405
+ * // Using Zod
317
406
  * z.object({
318
407
  * temperature: z.number(),
319
408
  * conditions: z.string(),
@@ -473,21 +562,72 @@ export type AgentLoopStrategy = (state: AgentLoopState) => boolean
473
562
  /**
474
563
  * Options passed into the SDK and further piped to the AI provider.
475
564
  */
476
- export interface ChatOptions<
477
- TModel extends string = string,
565
+ export interface TextOptions<
478
566
  TProviderOptionsSuperset extends Record<string, any> = Record<string, any>,
479
- TOutput extends ResponseFormat<any> | undefined = undefined,
480
567
  TProviderOptionsForModel = TProviderOptionsSuperset,
481
568
  > {
482
- model: TModel
569
+ model: string
483
570
  messages: Array<ModelMessage>
484
- tools?: Array<Tool>
571
+ tools?: Array<Tool<any, any, any>>
485
572
  systemPrompts?: Array<string>
486
573
  agentLoopStrategy?: AgentLoopStrategy
487
- options?: CommonOptions
488
- providerOptions?: TProviderOptionsForModel
574
+ /**
575
+ * Controls the randomness of the output.
576
+ * Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
577
+ * Range: [0.0, 2.0]
578
+ *
579
+ * Note: Generally recommended to use either temperature or topP, but not both.
580
+ *
581
+ * Provider usage:
582
+ * - OpenAI: `temperature` (number) - in text.top_p field
583
+ * - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
584
+ * - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
585
+ */
586
+ temperature?: number
587
+ /**
588
+ * Nucleus sampling parameter. An alternative to temperature sampling.
589
+ * The model considers the results of tokens with topP probability mass.
590
+ * For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
591
+ *
592
+ * Note: Generally recommended to use either temperature or topP, but not both.
593
+ *
594
+ * Provider usage:
595
+ * - OpenAI: `text.top_p` (number)
596
+ * - Anthropic: `top_p` (number | null)
597
+ * - Gemini: `generationConfig.topP` (number)
598
+ */
599
+ topP?: number
600
+ /**
601
+ * The maximum number of tokens to generate in the response.
602
+ *
603
+ * Provider usage:
604
+ * - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
605
+ * - Anthropic: `max_tokens` (number, required) - range x >= 1
606
+ * - Gemini: `generationConfig.maxOutputTokens` (number)
607
+ */
608
+ maxTokens?: number
609
+ /**
610
+ * Additional metadata to attach to the request.
611
+ * Can be used for tracking, debugging, or passing custom information.
612
+ * Structure and constraints vary by provider.
613
+ *
614
+ * Provider usage:
615
+ * - OpenAI: `metadata` (Record<string, string>) - max 16 key-value pairs, keys max 64 chars, values max 512 chars
616
+ * - Anthropic: `metadata` (Record<string, any>) - includes optional user_id (max 256 chars)
617
+ * - Gemini: Not directly available in TextProviderOptions
618
+ */
619
+ metadata?: Record<string, any>
620
+ modelOptions?: TProviderOptionsForModel
489
621
  request?: Request | RequestInit
490
- output?: TOutput
622
+
623
+ /**
624
+ * Schema for structured output.
625
+ * When provided, the adapter should use the provider's native structured output API
626
+ * to ensure the response conforms to this schema.
627
+ * The schema will be converted to JSON Schema format before being sent to the provider.
628
+ * Supports any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.).
629
+ */
630
+ outputSchema?: SchemaInput
491
631
  /**
492
632
  * Conversation ID for correlating client and server-side devtools events.
493
633
  * When provided, server-side events will be linked to the client conversation in devtools.
@@ -607,9 +747,9 @@ export type StreamChunk =
607
747
  | ToolInputAvailableStreamChunk
608
748
  | ThinkingStreamChunk
609
749
 
610
- // Simple streaming format for basic chat completions
611
- // Converted to StreamChunk format by convertChatCompletionStream()
612
- export interface ChatCompletionChunk {
750
+ // Simple streaming format for basic text completions
751
+ // Converted to StreamChunk format by convertTextCompletionStream()
752
+ export interface TextCompletionChunk {
613
753
  id: string
614
754
  model: string
615
755
  content: string
@@ -641,245 +781,250 @@ export interface SummarizationResult {
641
781
  }
642
782
  }
643
783
 
644
- export interface EmbeddingOptions {
784
+ // ============================================================================
785
+ // Image Generation Types
786
+ // ============================================================================
787
+
788
+ /**
789
+ * Options for image generation.
790
+ * These are the common options supported across providers.
791
+ */
792
+ export interface ImageGenerationOptions<
793
+ TProviderOptions extends object = object,
794
+ > {
795
+ /** The model to use for image generation */
645
796
  model: string
646
- input: string | Array<string>
647
- dimensions?: number
797
+ /** Text description of the desired image(s) */
798
+ prompt: string
799
+ /** Number of images to generate (default: 1) */
800
+ numberOfImages?: number
801
+ /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
802
+ size?: string
803
+ /** Model-specific options for image generation */
804
+ modelOptions?: TProviderOptions
648
805
  }
649
806
 
650
- export interface EmbeddingResult {
807
+ /**
808
+ * A single generated image
809
+ */
810
+ export interface GeneratedImage {
811
+ /** Base64-encoded image data */
812
+ b64Json?: string
813
+ /** URL to the generated image (may be temporary) */
814
+ url?: string
815
+ /** Revised prompt used by the model (if applicable) */
816
+ revisedPrompt?: string
817
+ }
818
+
819
+ /**
820
+ * Result of image generation
821
+ */
822
+ export interface ImageGenerationResult {
823
+ /** Unique identifier for the generation */
651
824
  id: string
825
+ /** Model used for generation */
652
826
  model: string
653
- embeddings: Array<Array<number>>
654
- usage: {
655
- promptTokens: number
656
- totalTokens: number
827
+ /** Array of generated images */
828
+ images: Array<GeneratedImage>
829
+ /** Token usage information (if available) */
830
+ usage?: {
831
+ inputTokens?: number
832
+ outputTokens?: number
833
+ totalTokens?: number
657
834
  }
658
835
  }
659
836
 
837
+ // ============================================================================
838
+ // Video Generation Types (Experimental)
839
+ // ============================================================================
840
+
660
841
  /**
661
- * Default metadata type for adapters that don't define custom metadata.
662
- * Uses unknown for all modalities.
842
+ * Options for video generation.
843
+ * These are the common options supported across providers.
844
+ *
845
+ * @experimental Video generation is an experimental feature and may change.
663
846
  */
664
- export interface DefaultMessageMetadataByModality {
665
- text: unknown
666
- image: unknown
667
- audio: unknown
668
- video: unknown
669
- document: unknown
847
+ export interface VideoGenerationOptions<
848
+ TProviderOptions extends object = object,
849
+ > {
850
+ /** The model to use for video generation */
851
+ model: string
852
+ /** Text description of the desired video */
853
+ prompt: string
854
+ /** Video size in WIDTHxHEIGHT format (e.g., "1280x720") */
855
+ size?: string
856
+ /** Video duration in seconds */
857
+ duration?: number
858
+ /** Model-specific options for video generation */
859
+ modelOptions?: TProviderOptions
670
860
  }
671
861
 
672
862
  /**
673
- * AI adapter interface with support for endpoint-specific models and provider options.
863
+ * Result of creating a video generation job.
674
864
  *
675
- * Generic parameters:
676
- * - TChatModels: Models that support chat/text completion
677
- * - TEmbeddingModels: Models that support embeddings
678
- * - TChatProviderOptions: Provider-specific options for chat endpoint
679
- * - TEmbeddingProviderOptions: Provider-specific options for embedding endpoint
680
- * - TModelProviderOptionsByName: Map from model name to its specific provider options
681
- * - TModelInputModalitiesByName: Map from model name to its supported input modalities
682
- * - TMessageMetadataByModality: Map from modality type to adapter-specific metadata types
683
- */
684
- export interface AIAdapter<
685
- TChatModels extends ReadonlyArray<string> = ReadonlyArray<string>,
686
- TEmbeddingModels extends ReadonlyArray<string> = ReadonlyArray<string>,
687
- TChatProviderOptions extends Record<string, any> = Record<string, any>,
688
- TEmbeddingProviderOptions extends Record<string, any> = Record<string, any>,
689
- TModelProviderOptionsByName extends Record<string, any> = Record<string, any>,
690
- TModelInputModalitiesByName extends Record<
691
- string,
692
- ReadonlyArray<Modality>
693
- > = Record<string, ReadonlyArray<Modality>>,
694
- TMessageMetadataByModality extends {
695
- text: unknown
696
- image: unknown
697
- audio: unknown
698
- video: unknown
699
- document: unknown
700
- } = DefaultMessageMetadataByModality,
701
- > {
702
- name: string
703
- /** Models that support chat/text completion */
704
- models: TChatModels
865
+ * @experimental Video generation is an experimental feature and may change.
866
+ */
867
+ export interface VideoJobResult {
868
+ /** Unique job identifier for polling status */
869
+ jobId: string
870
+ /** Model used for generation */
871
+ model: string
872
+ }
705
873
 
706
- /** Models that support embeddings */
707
- embeddingModels?: TEmbeddingModels
874
+ /**
875
+ * Status of a video generation job.
876
+ *
877
+ * @experimental Video generation is an experimental feature and may change.
878
+ */
879
+ export interface VideoStatusResult {
880
+ /** Job identifier */
881
+ jobId: string
882
+ /** Current status of the job */
883
+ status: 'pending' | 'processing' | 'completed' | 'failed'
884
+ /** Progress percentage (0-100), if available */
885
+ progress?: number
886
+ /** Error message if status is 'failed' */
887
+ error?: string
888
+ }
708
889
 
709
- // Type-only properties for provider options inference
710
- _providerOptions?: TChatProviderOptions // Alias for _chatProviderOptions
711
- _chatProviderOptions?: TChatProviderOptions
712
- _embeddingProviderOptions?: TEmbeddingProviderOptions
713
- /**
714
- * Type-only map from model name to its specific provider options.
715
- * Used by the core AI types to narrow providerOptions based on the selected model.
716
- * Must be provided by all adapters.
717
- */
718
- _modelProviderOptionsByName: TModelProviderOptionsByName
719
- /**
720
- * Type-only map from model name to its supported input modalities.
721
- * Used by the core AI types to narrow ContentPart types based on the selected model.
722
- * Must be provided by all adapters.
723
- */
724
- _modelInputModalitiesByName?: TModelInputModalitiesByName
725
- /**
726
- * Type-only map from modality type to adapter-specific metadata types.
727
- * Used to provide type-safe autocomplete for metadata on content parts.
728
- */
729
- _messageMetadataByModality?: TMessageMetadataByModality
890
+ /**
891
+ * Result containing the URL to a generated video.
892
+ *
893
+ * @experimental Video generation is an experimental feature and may change.
894
+ */
895
+ export interface VideoUrlResult {
896
+ /** Job identifier */
897
+ jobId: string
898
+ /** URL to the generated video */
899
+ url: string
900
+ /** When the URL expires, if applicable */
901
+ expiresAt?: Date
902
+ }
730
903
 
731
- // Structured streaming with JSON chunks (supports tool calls and rich content)
732
- chatStream: (
733
- options: ChatOptions<string, TChatProviderOptions>,
734
- ) => AsyncIterable<StreamChunk>
904
+ // ============================================================================
905
+ // Text-to-Speech (TTS) Types
906
+ // ============================================================================
735
907
 
736
- // Summarization
737
- summarize: (options: SummarizationOptions) => Promise<SummarizationResult>
908
+ /**
909
+ * Options for text-to-speech generation.
910
+ * These are the common options supported across providers.
911
+ */
912
+ export interface TTSOptions<TProviderOptions extends object = object> {
913
+ /** The model to use for TTS generation */
914
+ model: string
915
+ /** The text to convert to speech */
916
+ text: string
917
+ /** The voice to use for generation */
918
+ voice?: string
919
+ /** The output audio format */
920
+ format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'
921
+ /** The speed of the generated audio (0.25 to 4.0) */
922
+ speed?: number
923
+ /** Model-specific options for TTS generation */
924
+ modelOptions?: TProviderOptions
925
+ }
738
926
 
739
- // Embeddings
740
- createEmbeddings: (options: EmbeddingOptions) => Promise<EmbeddingResult>
927
+ /**
928
+ * Result of text-to-speech generation.
929
+ */
930
+ export interface TTSResult {
931
+ /** Unique identifier for the generation */
932
+ id: string
933
+ /** Model used for generation */
934
+ model: string
935
+ /** Base64-encoded audio data */
936
+ audio: string
937
+ /** Audio format of the generated audio */
938
+ format: string
939
+ /** Duration of the audio in seconds, if available */
940
+ duration?: number
941
+ /** Content type of the audio (e.g., 'audio/mp3') */
942
+ contentType?: string
943
+ }
944
+
945
+ // ============================================================================
946
+ // Transcription (Speech-to-Text) Types
947
+ // ============================================================================
948
+
949
+ /**
950
+ * Options for audio transcription.
951
+ * These are the common options supported across providers.
952
+ */
953
+ export interface TranscriptionOptions<
954
+ TProviderOptions extends object = object,
955
+ > {
956
+ /** The model to use for transcription */
957
+ model: string
958
+ /** The audio data to transcribe - can be base64 string, File, Blob, or Buffer */
959
+ audio: string | File | Blob | ArrayBuffer
960
+ /** The language of the audio in ISO-639-1 format (e.g., 'en') */
961
+ language?: string
962
+ /** An optional prompt to guide the transcription */
963
+ prompt?: string
964
+ /** The format of the transcription output */
965
+ responseFormat?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt'
966
+ /** Model-specific options for transcription */
967
+ modelOptions?: TProviderOptions
741
968
  }
742
969
 
743
- export interface AIAdapterConfig {
744
- apiKey?: string
745
- baseUrl?: string
746
- timeout?: number
747
- maxRetries?: number
748
- headers?: Record<string, string>
970
+ /**
971
+ * A single segment of transcribed audio with timing information.
972
+ */
973
+ export interface TranscriptionSegment {
974
+ /** Unique identifier for the segment */
975
+ id: number
976
+ /** Start time of the segment in seconds */
977
+ start: number
978
+ /** End time of the segment in seconds */
979
+ end: number
980
+ /** Transcribed text for this segment */
981
+ text: string
982
+ /** Confidence score (0-1), if available */
983
+ confidence?: number
984
+ /** Speaker identifier, if diarization is enabled */
985
+ speaker?: string
749
986
  }
750
987
 
751
- export type ChatStreamOptionsUnion<
752
- TAdapter extends AIAdapter<any, any, any, any, any, any, any>,
753
- > =
754
- TAdapter extends AIAdapter<
755
- infer Models,
756
- any,
757
- any,
758
- any,
759
- infer ModelProviderOptions,
760
- infer ModelInputModalities,
761
- infer MessageMetadata
762
- >
763
- ? Models[number] extends infer TModel
764
- ? TModel extends string
765
- ? Omit<
766
- ChatOptions,
767
- 'model' | 'providerOptions' | 'responseFormat' | 'messages'
768
- > & {
769
- adapter: TAdapter
770
- model: TModel
771
- providerOptions?: TModel extends keyof ModelProviderOptions
772
- ? ModelProviderOptions[TModel]
773
- : never
774
- /**
775
- * Messages array with content constrained to the model's supported input modalities.
776
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
777
- * Metadata types are also constrained based on the adapter's metadata type definitions.
778
- */
779
- messages: TModel extends keyof ModelInputModalities
780
- ? ModelInputModalities[TModel] extends ReadonlyArray<Modality>
781
- ? MessageMetadata extends {
782
- text: infer TTextMeta
783
- image: infer TImageMeta
784
- audio: infer TAudioMeta
785
- video: infer TVideoMeta
786
- document: infer TDocumentMeta
787
- }
788
- ? Array<
789
- ConstrainedModelMessage<
790
- ModelInputModalities[TModel],
791
- TImageMeta,
792
- TAudioMeta,
793
- TVideoMeta,
794
- TDocumentMeta,
795
- TTextMeta
796
- >
797
- >
798
- : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>>
799
- : Array<ModelMessage>
800
- : Array<ModelMessage>
801
- }
802
- : never
803
- : never
804
- : never
805
-
806
- /**
807
- * Chat options constrained by a specific model's capabilities.
808
- * Unlike ChatStreamOptionsUnion which creates a union over all models,
809
- * this type takes a specific model and constrains messages accordingly.
810
- */
811
- export type ChatStreamOptionsForModel<
812
- TAdapter extends AIAdapter<any, any, any, any, any, any, any>,
813
- TModel extends string,
814
- > =
815
- TAdapter extends AIAdapter<
816
- any,
817
- any,
818
- any,
819
- any,
820
- infer ModelProviderOptions,
821
- infer ModelInputModalities,
822
- infer MessageMetadata
823
- >
824
- ? Omit<
825
- ChatOptions,
826
- 'model' | 'providerOptions' | 'responseFormat' | 'messages'
827
- > & {
828
- adapter: TAdapter
829
- model: TModel
830
- providerOptions?: TModel extends keyof ModelProviderOptions
831
- ? ModelProviderOptions[TModel]
832
- : never
833
- /**
834
- * Messages array with content constrained to the model's supported input modalities.
835
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
836
- * Metadata types are also constrained based on the adapter's metadata type definitions.
837
- */
838
- messages: TModel extends keyof ModelInputModalities
839
- ? ModelInputModalities[TModel] extends ReadonlyArray<Modality>
840
- ? MessageMetadata extends {
841
- text: infer TTextMeta
842
- image: infer TImageMeta
843
- audio: infer TAudioMeta
844
- video: infer TVideoMeta
845
- document: infer TDocumentMeta
846
- }
847
- ? Array<
848
- ConstrainedModelMessage<
849
- ModelInputModalities[TModel],
850
- TImageMeta,
851
- TAudioMeta,
852
- TVideoMeta,
853
- TDocumentMeta,
854
- TTextMeta
855
- >
856
- >
857
- : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>>
858
- : Array<ModelMessage>
859
- : Array<ModelMessage>
860
- }
861
- : never
862
-
863
- // Extract types from adapter (updated to 6 generics)
864
- export type ExtractModelsFromAdapter<T> =
865
- T extends AIAdapter<infer M, any, any, any, any, any> ? M[number] : never
866
-
867
- /**
868
- * Extract the supported input modalities for a specific model from an adapter.
869
- */
870
- export type ExtractModalitiesForModel<
871
- TAdapter extends AIAdapter<any, any, any, any, any, any>,
872
- TModel extends string,
873
- > =
874
- TAdapter extends AIAdapter<
875
- any,
876
- any,
877
- any,
878
- any,
879
- any,
880
- infer ModelInputModalities
881
- >
882
- ? TModel extends keyof ModelInputModalities
883
- ? ModelInputModalities[TModel]
884
- : ReadonlyArray<Modality>
885
- : ReadonlyArray<Modality>
988
+ /**
989
+ * A single word with timing information.
990
+ */
991
+ export interface TranscriptionWord {
992
+ /** The transcribed word */
993
+ word: string
994
+ /** Start time in seconds */
995
+ start: number
996
+ /** End time in seconds */
997
+ end: number
998
+ }
999
+
1000
+ /**
1001
+ * Result of audio transcription.
1002
+ */
1003
+ export interface TranscriptionResult {
1004
+ /** Unique identifier for the transcription */
1005
+ id: string
1006
+ /** Model used for transcription */
1007
+ model: string
1008
+ /** The full transcribed text */
1009
+ text: string
1010
+ /** Language detected or specified */
1011
+ language?: string
1012
+ /** Duration of the audio in seconds */
1013
+ duration?: number
1014
+ /** Detailed segments with timing, if available */
1015
+ segments?: Array<TranscriptionSegment>
1016
+ /** Word-level timestamps, if available */
1017
+ words?: Array<TranscriptionWord>
1018
+ }
1019
+
1020
+ /**
1021
+ * Default metadata type for adapters that don't define custom metadata.
1022
+ * Uses unknown for all modalities.
1023
+ */
1024
+ export interface DefaultMessageMetadataByModality {
1025
+ text: unknown
1026
+ image: unknown
1027
+ audio: unknown
1028
+ video: unknown
1029
+ document: unknown
1030
+ }