@tanstack/ai 0.0.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (146) hide show
  1. package/README.md +26 -0
  2. package/dist/esm/activities/chat/adapter.d.ts +100 -0
  3. package/dist/esm/activities/chat/adapter.js +14 -0
  4. package/dist/esm/activities/chat/adapter.js.map +1 -0
  5. package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.d.ts +4 -4
  6. package/dist/esm/activities/chat/agent-loop-strategies.js.map +1 -0
  7. package/dist/esm/activities/chat/index.d.ts +166 -0
  8. package/dist/esm/{core/chat.js → activities/chat/index.js} +131 -33
  9. package/dist/esm/activities/chat/index.js.map +1 -0
  10. package/dist/esm/{message-converters.d.ts → activities/chat/messages.d.ts} +1 -1
  11. package/dist/esm/{message-converters.js → activities/chat/messages.js} +7 -7
  12. package/dist/esm/activities/chat/messages.js.map +1 -0
  13. package/dist/esm/activities/chat/stream/json-parser.js.map +1 -0
  14. package/dist/esm/{stream → activities/chat/stream}/message-updaters.d.ts +1 -1
  15. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -0
  16. package/dist/esm/{stream → activities/chat/stream}/processor.d.ts +1 -1
  17. package/dist/esm/{stream → activities/chat/stream}/processor.js +1 -1
  18. package/dist/esm/activities/chat/stream/processor.js.map +1 -0
  19. package/dist/esm/activities/chat/stream/strategies.js.map +1 -0
  20. package/dist/esm/{stream → activities/chat/stream}/types.d.ts +2 -9
  21. package/dist/esm/activities/chat/tools/schema-converter.d.ts +116 -0
  22. package/dist/esm/activities/chat/tools/schema-converter.js +115 -0
  23. package/dist/esm/activities/chat/tools/schema-converter.js.map +1 -0
  24. package/dist/esm/{tools → activities/chat/tools}/tool-calls.d.ts +1 -1
  25. package/dist/esm/{tools → activities/chat/tools}/tool-calls.js +23 -30
  26. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -0
  27. package/dist/esm/{tools → activities/chat/tools}/tool-definition.d.ts +22 -18
  28. package/dist/esm/activities/chat/tools/tool-definition.js.map +1 -0
  29. package/dist/esm/activities/generateImage/adapter.d.ts +68 -0
  30. package/dist/esm/activities/generateImage/adapter.js +14 -0
  31. package/dist/esm/activities/generateImage/adapter.js.map +1 -0
  32. package/dist/esm/activities/generateImage/index.d.ts +89 -0
  33. package/dist/esm/activities/generateImage/index.js +15 -0
  34. package/dist/esm/activities/generateImage/index.js.map +1 -0
  35. package/dist/esm/activities/generateSpeech/adapter.d.ts +62 -0
  36. package/dist/esm/activities/generateSpeech/adapter.js +14 -0
  37. package/dist/esm/activities/generateSpeech/adapter.js.map +1 -0
  38. package/dist/esm/activities/generateSpeech/index.d.ts +69 -0
  39. package/dist/esm/activities/generateSpeech/index.js +15 -0
  40. package/dist/esm/activities/generateSpeech/index.js.map +1 -0
  41. package/dist/esm/activities/generateTranscription/adapter.d.ts +62 -0
  42. package/dist/esm/activities/generateTranscription/adapter.js +14 -0
  43. package/dist/esm/activities/generateTranscription/adapter.js.map +1 -0
  44. package/dist/esm/activities/generateTranscription/index.d.ts +71 -0
  45. package/dist/esm/activities/generateTranscription/index.js +15 -0
  46. package/dist/esm/activities/generateTranscription/index.js.map +1 -0
  47. package/dist/esm/activities/generateVideo/adapter.d.ts +80 -0
  48. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  49. package/dist/esm/activities/generateVideo/adapter.js.map +1 -0
  50. package/dist/esm/activities/generateVideo/index.d.ts +136 -0
  51. package/dist/esm/activities/generateVideo/index.js +47 -0
  52. package/dist/esm/activities/generateVideo/index.js.map +1 -0
  53. package/dist/esm/activities/index.d.ts +22 -0
  54. package/dist/esm/activities/index.js +34 -0
  55. package/dist/esm/activities/index.js.map +1 -0
  56. package/dist/esm/activities/summarize/adapter.d.ts +74 -0
  57. package/dist/esm/activities/summarize/adapter.js +14 -0
  58. package/dist/esm/activities/summarize/adapter.js.map +1 -0
  59. package/dist/esm/activities/summarize/index.d.ts +100 -0
  60. package/dist/esm/activities/summarize/index.js +90 -0
  61. package/dist/esm/activities/summarize/index.js.map +1 -0
  62. package/dist/esm/event-client.d.ts +4 -18
  63. package/dist/esm/event-client.js.map +1 -1
  64. package/dist/esm/index.d.ts +16 -14
  65. package/dist/esm/index.js +29 -20
  66. package/dist/esm/stream-to-response.d.ts +95 -0
  67. package/dist/esm/stream-to-response.js +118 -0
  68. package/dist/esm/stream-to-response.js.map +1 -0
  69. package/dist/esm/types.d.ts +370 -133
  70. package/package.json +7 -6
  71. package/src/activities/chat/adapter.ts +150 -0
  72. package/src/{utilities → activities/chat}/agent-loop-strategies.ts +4 -4
  73. package/src/{core/chat.ts → activities/chat/index.ts} +435 -79
  74. package/src/{message-converters.ts → activities/chat/messages.ts} +10 -13
  75. package/src/{stream → activities/chat/stream}/message-updaters.ts +1 -1
  76. package/src/{stream → activities/chat/stream}/processor.ts +2 -5
  77. package/src/{stream → activities/chat/stream}/types.ts +8 -18
  78. package/src/activities/chat/tools/schema-converter.ts +332 -0
  79. package/src/{tools → activities/chat/tools}/tool-calls.ts +63 -44
  80. package/src/{tools → activities/chat/tools}/tool-definition.ts +51 -38
  81. package/src/activities/generateImage/adapter.ts +104 -0
  82. package/src/activities/generateImage/index.ts +162 -0
  83. package/src/activities/generateSpeech/adapter.ts +87 -0
  84. package/src/activities/generateSpeech/index.ts +122 -0
  85. package/src/activities/generateTranscription/adapter.ts +89 -0
  86. package/src/activities/generateTranscription/index.ts +132 -0
  87. package/src/activities/generateVideo/adapter.ts +116 -0
  88. package/src/activities/generateVideo/index.ts +261 -0
  89. package/src/activities/index.ts +164 -0
  90. package/src/activities/summarize/adapter.ts +107 -0
  91. package/src/activities/summarize/index.ts +287 -0
  92. package/src/event-client.ts +5 -21
  93. package/src/index.ts +60 -15
  94. package/src/stream-to-response.ts +237 -0
  95. package/src/types.ts +429 -284
  96. package/dist/esm/base-adapter.d.ts +0 -36
  97. package/dist/esm/base-adapter.js +0 -12
  98. package/dist/esm/base-adapter.js.map +0 -1
  99. package/dist/esm/core/chat-common-options.d.ts +0 -52
  100. package/dist/esm/core/chat.d.ts +0 -30
  101. package/dist/esm/core/chat.js.map +0 -1
  102. package/dist/esm/core/embedding.d.ts +0 -8
  103. package/dist/esm/core/embedding.js +0 -33
  104. package/dist/esm/core/embedding.js.map +0 -1
  105. package/dist/esm/core/summarize.d.ts +0 -9
  106. package/dist/esm/core/summarize.js +0 -36
  107. package/dist/esm/core/summarize.js.map +0 -1
  108. package/dist/esm/message-converters.js.map +0 -1
  109. package/dist/esm/stream/json-parser.js.map +0 -1
  110. package/dist/esm/stream/message-updaters.js.map +0 -1
  111. package/dist/esm/stream/processor.js.map +0 -1
  112. package/dist/esm/stream/strategies.js.map +0 -1
  113. package/dist/esm/tools/tool-calls.js.map +0 -1
  114. package/dist/esm/tools/tool-definition.js.map +0 -1
  115. package/dist/esm/tools/zod-converter.d.ts +0 -30
  116. package/dist/esm/tools/zod-converter.js +0 -36
  117. package/dist/esm/tools/zod-converter.js.map +0 -1
  118. package/dist/esm/utilities/agent-loop-strategies.js.map +0 -1
  119. package/dist/esm/utilities/chat-options.d.ts +0 -6
  120. package/dist/esm/utilities/chat-options.js +0 -7
  121. package/dist/esm/utilities/chat-options.js.map +0 -1
  122. package/dist/esm/utilities/messages.d.ts +0 -30
  123. package/dist/esm/utilities/messages.js +0 -7
  124. package/dist/esm/utilities/messages.js.map +0 -1
  125. package/dist/esm/utilities/stream-to-response.d.ts +0 -48
  126. package/dist/esm/utilities/stream-to-response.js +0 -62
  127. package/dist/esm/utilities/stream-to-response.js.map +0 -1
  128. package/src/base-adapter.ts +0 -86
  129. package/src/core/chat-common-options.ts +0 -55
  130. package/src/core/embedding.ts +0 -54
  131. package/src/core/summarize.ts +0 -56
  132. package/src/tools/zod-converter.ts +0 -85
  133. package/src/utilities/chat-options.ts +0 -35
  134. package/src/utilities/messages.ts +0 -63
  135. package/src/utilities/stream-to-response.ts +0 -116
  136. /package/dist/esm/{utilities → activities/chat}/agent-loop-strategies.js +0 -0
  137. /package/dist/esm/{stream → activities/chat/stream}/index.d.ts +0 -0
  138. /package/dist/esm/{stream → activities/chat/stream}/json-parser.d.ts +0 -0
  139. /package/dist/esm/{stream → activities/chat/stream}/json-parser.js +0 -0
  140. /package/dist/esm/{stream → activities/chat/stream}/message-updaters.js +0 -0
  141. /package/dist/esm/{stream → activities/chat/stream}/strategies.d.ts +0 -0
  142. /package/dist/esm/{stream → activities/chat/stream}/strategies.js +0 -0
  143. /package/dist/esm/{tools → activities/chat/tools}/tool-definition.js +0 -0
  144. /package/src/{stream → activities/chat/stream}/index.ts +0 -0
  145. /package/src/{stream → activities/chat/stream}/json-parser.ts +0 -0
  146. /package/src/{stream → activities/chat/stream}/strategies.ts +0 -0
@@ -1,6 +1,73 @@
1
- import { CommonOptions } from './core/chat-common-options.js';
2
- import { z } from 'zod';
3
- import { ToolCallState, ToolResultState } from './stream/types.js';
1
+ import { StandardJSONSchemaV1 } from '@standard-schema/spec';
2
+ /**
3
+ * Tool call states - track the lifecycle of a tool call
4
+ */
5
+ export type ToolCallState = 'awaiting-input' | 'input-streaming' | 'input-complete' | 'approval-requested' | 'approval-responded';
6
+ /**
7
+ * Tool result states - track the lifecycle of a tool result
8
+ */
9
+ export type ToolResultState = 'streaming' | 'complete' | 'error';
10
+ /**
11
+ * JSON Schema type for defining tool input/output schemas as raw JSON Schema objects.
12
+ * This allows tools to be defined without schema libraries when you have JSON Schema definitions available.
13
+ */
14
+ export interface JSONSchema {
15
+ type?: string | Array<string>;
16
+ properties?: Record<string, JSONSchema>;
17
+ items?: JSONSchema | Array<JSONSchema>;
18
+ required?: Array<string>;
19
+ enum?: Array<unknown>;
20
+ const?: unknown;
21
+ description?: string;
22
+ default?: unknown;
23
+ $ref?: string;
24
+ $defs?: Record<string, JSONSchema>;
25
+ definitions?: Record<string, JSONSchema>;
26
+ allOf?: Array<JSONSchema>;
27
+ anyOf?: Array<JSONSchema>;
28
+ oneOf?: Array<JSONSchema>;
29
+ not?: JSONSchema;
30
+ if?: JSONSchema;
31
+ then?: JSONSchema;
32
+ else?: JSONSchema;
33
+ minimum?: number;
34
+ maximum?: number;
35
+ exclusiveMinimum?: number;
36
+ exclusiveMaximum?: number;
37
+ minLength?: number;
38
+ maxLength?: number;
39
+ pattern?: string;
40
+ format?: string;
41
+ minItems?: number;
42
+ maxItems?: number;
43
+ uniqueItems?: boolean;
44
+ additionalProperties?: boolean | JSONSchema;
45
+ additionalItems?: boolean | JSONSchema;
46
+ patternProperties?: Record<string, JSONSchema>;
47
+ propertyNames?: JSONSchema;
48
+ minProperties?: number;
49
+ maxProperties?: number;
50
+ title?: string;
51
+ examples?: Array<unknown>;
52
+ [key: string]: unknown;
53
+ }
54
+ /**
55
+ * Union type for schema input - can be any Standard JSON Schema compliant schema or a plain JSONSchema object.
56
+ *
57
+ * Standard JSON Schema compliant libraries include:
58
+ * - Zod v4.2+ (natively supports StandardJSONSchemaV1)
59
+ * - ArkType v2.1.28+ (natively supports StandardJSONSchemaV1)
60
+ * - Valibot v1.2+ (via `toStandardJsonSchema()` from `@valibot/to-json-schema`)
61
+ *
62
+ * @see https://standardschema.dev/json-schema
63
+ */
64
+ export type SchemaInput = StandardJSONSchemaV1<any, any> | JSONSchema;
65
+ /**
66
+ * Infer the TypeScript type from a schema.
67
+ * For Standard JSON Schema compliant schemas, extracts the input type.
68
+ * For plain JSONSchema, returns `any` since we can't infer types from JSON Schema at compile time.
69
+ */
70
+ export type InferSchemaType<T> = T extends StandardJSONSchemaV1<infer TInput, unknown> ? TInput : unknown;
4
71
  export interface ToolCall {
5
72
  id: string;
6
73
  type: 'function';
@@ -87,13 +154,13 @@ export interface DocumentPart<TMetadata = unknown> {
87
154
  * @template TVideoMeta - Provider-specific video metadata type
88
155
  * @template TDocumentMeta - Provider-specific document metadata type
89
156
  */
90
- export type ContentPart<TImageMeta = unknown, TAudioMeta = unknown, TVideoMeta = unknown, TDocumentMeta = unknown, TTextMeta = unknown> = TextPart<TTextMeta> | ImagePart<TImageMeta> | AudioPart<TAudioMeta> | VideoPart<TVideoMeta> | DocumentPart<TDocumentMeta>;
157
+ export type ContentPart<TTextMeta = unknown, TImageMeta = unknown, TAudioMeta = unknown, TVideoMeta = unknown, TDocumentMeta = unknown> = TextPart<TTextMeta> | ImagePart<TImageMeta> | AudioPart<TAudioMeta> | VideoPart<TVideoMeta> | DocumentPart<TDocumentMeta>;
91
158
  /**
92
159
  * Helper type to filter ContentPart union to only include specific modalities.
93
160
  * Used to constrain message content based on model capabilities.
94
161
  */
95
- export type ContentPartForModalities<TModalities extends Modality, TImageMeta = unknown, TAudioMeta = unknown, TVideoMeta = unknown, TDocumentMeta = unknown, TTextMeta = unknown> = Extract<ContentPart<TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>, {
96
- type: TModalities;
162
+ export type ContentPartForInputModalitiesTypes<TInputModalitiesTypes extends InputModalitiesTypes> = Extract<ContentPart<TInputModalitiesTypes['messageMetadataByModality']['text'], TInputModalitiesTypes['messageMetadataByModality']['image'], TInputModalitiesTypes['messageMetadataByModality']['audio'], TInputModalitiesTypes['messageMetadataByModality']['video'], TInputModalitiesTypes['messageMetadataByModality']['document']>, {
163
+ type: TInputModalitiesTypes['inputModalities'][number];
97
164
  }>;
98
165
  /**
99
166
  * Helper type to convert a readonly array of modalities to a union type.
@@ -104,7 +171,7 @@ export type ModalitiesArrayToUnion<T extends ReadonlyArray<Modality>> = T[number
104
171
  * Type for message content constrained by supported modalities.
105
172
  * When modalities is ['text', 'image'], only TextPart and ImagePart are allowed in the array.
106
173
  */
107
- export type ConstrainedContent<TModalities extends ReadonlyArray<Modality>, TImageMeta = unknown, TAudioMeta = unknown, TVideoMeta = unknown, TDocumentMeta = unknown, TTextMeta = unknown> = string | null | Array<ContentPartForModalities<ModalitiesArrayToUnion<TModalities>, TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>>;
174
+ export type ConstrainedContent<TInputModalitiesTypes extends InputModalitiesTypes> = string | null | Array<ContentPartForInputModalitiesTypes<TInputModalitiesTypes>>;
108
175
  export interface ModelMessage<TContent extends string | null | Array<ContentPart> = string | null | Array<ContentPart>> {
109
176
  role: 'user' | 'assistant' | 'tool';
110
177
  content: TContent;
@@ -157,12 +224,16 @@ export interface UIMessage {
157
224
  parts: Array<MessagePart>;
158
225
  createdAt?: Date;
159
226
  }
227
+ export type InputModalitiesTypes = {
228
+ inputModalities: ReadonlyArray<Modality>;
229
+ messageMetadataByModality: DefaultMessageMetadataByModality;
230
+ };
160
231
  /**
161
232
  * A ModelMessage with content constrained to only allow content parts
162
233
  * matching the specified input modalities.
163
234
  */
164
- export type ConstrainedModelMessage<TModalities extends ReadonlyArray<Modality>, TImageMeta = unknown, TAudioMeta = unknown, TVideoMeta = unknown, TDocumentMeta = unknown, TTextMeta = unknown> = Omit<ModelMessage, 'content'> & {
165
- content: ConstrainedContent<TModalities, TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>;
235
+ export type ConstrainedModelMessage<TInputModalitiesTypes extends InputModalitiesTypes> = Omit<ModelMessage, 'content'> & {
236
+ content: ConstrainedContent<TInputModalitiesTypes>;
166
237
  };
167
238
  /**
168
239
  * Tool/Function definition for function calling.
@@ -170,12 +241,14 @@ export type ConstrainedModelMessage<TModalities extends ReadonlyArray<Modality>,
170
241
  * Tools allow the model to interact with external systems, APIs, or perform computations.
171
242
  * The model will decide when to call tools based on the user's request and the tool descriptions.
172
243
  *
173
- * Tools use Zod schemas for runtime validation and type safety.
244
+ * Tools can use any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.)
245
+ * or plain JSON Schema objects for runtime validation and type safety.
174
246
  *
175
247
  * @see https://platform.openai.com/docs/guides/function-calling
176
248
  * @see https://docs.anthropic.com/claude/docs/tool-use
249
+ * @see https://standardschema.dev/json-schema
177
250
  */
178
- export interface Tool<TInput extends z.ZodType = z.ZodType, TOutput extends z.ZodType = z.ZodType, TName extends string = string> {
251
+ export interface Tool<TInput extends SchemaInput = SchemaInput, TOutput extends SchemaInput = SchemaInput, TName extends string = string> {
179
252
  /**
180
253
  * Unique name of the tool (used by the model to call it).
181
254
  *
@@ -195,33 +268,57 @@ export interface Tool<TInput extends z.ZodType = z.ZodType, TOutput extends z.Zo
195
268
  */
196
269
  description: string;
197
270
  /**
198
- * Zod schema describing the tool's input parameters.
271
+ * Schema describing the tool's input parameters.
199
272
  *
273
+ * Can be any Standard JSON Schema compliant schema (Zod, ArkType, Valibot, etc.) or a plain JSON Schema object.
200
274
  * Defines the structure and types of arguments the tool accepts.
201
275
  * The model will generate arguments matching this schema.
202
- * The schema is converted to JSON Schema for LLM providers.
276
+ * Standard JSON Schema compliant schemas are converted to JSON Schema for LLM providers.
203
277
  *
204
- * @see https://zod.dev/
278
+ * @see https://standardschema.dev/json-schema
279
+ * @see https://json-schema.org/
205
280
  *
206
281
  * @example
282
+ * // Using Zod v4+ schema (natively supports Standard JSON Schema)
207
283
  * import { z } from 'zod';
208
- *
209
284
  * z.object({
210
285
  * location: z.string().describe("City name or coordinates"),
211
286
  * unit: z.enum(["celsius", "fahrenheit"]).optional()
212
287
  * })
288
+ *
289
+ * @example
290
+ * // Using ArkType (natively supports Standard JSON Schema)
291
+ * import { type } from 'arktype';
292
+ * type({
293
+ * location: 'string',
294
+ * unit: "'celsius' | 'fahrenheit'"
295
+ * })
296
+ *
297
+ * @example
298
+ * // Using plain JSON Schema
299
+ * {
300
+ * type: 'object',
301
+ * properties: {
302
+ * location: { type: 'string', description: 'City name or coordinates' },
303
+ * unit: { type: 'string', enum: ['celsius', 'fahrenheit'] }
304
+ * },
305
+ * required: ['location']
306
+ * }
213
307
  */
214
308
  inputSchema?: TInput;
215
309
  /**
216
- * Optional Zod schema for validating tool output.
310
+ * Optional schema for validating tool output.
217
311
  *
218
- * If provided, tool results will be validated against this schema before
219
- * being sent back to the model. This catches bugs in tool implementations
220
- * and ensures consistent output formatting.
312
+ * Can be any Standard JSON Schema compliant schema or a plain JSON Schema object.
313
+ * If provided with a Standard Schema compliant schema, tool results will be validated
314
+ * against this schema before being sent back to the model. This catches bugs in tool
315
+ * implementations and ensures consistent output formatting.
221
316
  *
222
317
  * Note: This is client-side validation only - not sent to LLM providers.
318
+ * Note: Plain JSON Schema output validation is not performed at runtime.
223
319
  *
224
320
  * @example
321
+ * // Using Zod
225
322
  * z.object({
226
323
  * temperature: z.number(),
227
324
  * conditions: z.string(),
@@ -368,16 +465,68 @@ export type AgentLoopStrategy = (state: AgentLoopState) => boolean;
368
465
  /**
369
466
  * Options passed into the SDK and further piped to the AI provider.
370
467
  */
371
- export interface ChatOptions<TModel extends string = string, TProviderOptionsSuperset extends Record<string, any> = Record<string, any>, TOutput extends ResponseFormat<any> | undefined = undefined, TProviderOptionsForModel = TProviderOptionsSuperset> {
372
- model: TModel;
468
+ export interface TextOptions<TProviderOptionsSuperset extends Record<string, any> = Record<string, any>, TProviderOptionsForModel = TProviderOptionsSuperset> {
469
+ model: string;
373
470
  messages: Array<ModelMessage>;
374
- tools?: Array<Tool>;
471
+ tools?: Array<Tool<any, any, any>>;
375
472
  systemPrompts?: Array<string>;
376
473
  agentLoopStrategy?: AgentLoopStrategy;
377
- options?: CommonOptions;
378
- providerOptions?: TProviderOptionsForModel;
474
+ /**
475
+ * Controls the randomness of the output.
476
+ * Higher values (e.g., 0.8) make output more random, lower values (e.g., 0.2) make it more focused and deterministic.
477
+ * Range: [0.0, 2.0]
478
+ *
479
+ * Note: Generally recommended to use either temperature or topP, but not both.
480
+ *
481
+ * Provider usage:
482
+ * - OpenAI: `temperature` (number) - in text.top_p field
483
+ * - Anthropic: `temperature` (number) - ranges from 0.0 to 1.0, default 1.0
484
+ * - Gemini: `generationConfig.temperature` (number) - ranges from 0.0 to 2.0
485
+ */
486
+ temperature?: number;
487
+ /**
488
+ * Nucleus sampling parameter. An alternative to temperature sampling.
489
+ * The model considers the results of tokens with topP probability mass.
490
+ * For example, 0.1 means only tokens comprising the top 10% probability mass are considered.
491
+ *
492
+ * Note: Generally recommended to use either temperature or topP, but not both.
493
+ *
494
+ * Provider usage:
495
+ * - OpenAI: `text.top_p` (number)
496
+ * - Anthropic: `top_p` (number | null)
497
+ * - Gemini: `generationConfig.topP` (number)
498
+ */
499
+ topP?: number;
500
+ /**
501
+ * The maximum number of tokens to generate in the response.
502
+ *
503
+ * Provider usage:
504
+ * - OpenAI: `max_output_tokens` (number) - includes visible output and reasoning tokens
505
+ * - Anthropic: `max_tokens` (number, required) - range x >= 1
506
+ * - Gemini: `generationConfig.maxOutputTokens` (number)
507
+ */
508
+ maxTokens?: number;
509
+ /**
510
+ * Additional metadata to attach to the request.
511
+ * Can be used for tracking, debugging, or passing custom information.
512
+ * Structure and constraints vary by provider.
513
+ *
514
+ * Provider usage:
515
+ * - OpenAI: `metadata` (Record<string, string>) - max 16 key-value pairs, keys max 64 chars, values max 512 chars
516
+ * - Anthropic: `metadata` (Record<string, any>) - includes optional user_id (max 256 chars)
517
+ * - Gemini: Not directly available in TextProviderOptions
518
+ */
519
+ metadata?: Record<string, any>;
520
+ modelOptions?: TProviderOptionsForModel;
379
521
  request?: Request | RequestInit;
380
- output?: TOutput;
522
+ /**
523
+ * Schema for structured output.
524
+ * When provided, the adapter should use the provider's native structured output API
525
+ * to ensure the response conforms to this schema.
526
+ * The schema will be converted to JSON Schema format before being sent to the provider.
527
+ * Supports any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.).
528
+ */
529
+ outputSchema?: SchemaInput;
381
530
  /**
382
531
  * Conversation ID for correlating client and server-side devtools events.
383
532
  * When provided, server-side events will be linked to the client conversation in devtools.
@@ -469,7 +618,7 @@ export interface ThinkingStreamChunk extends BaseStreamChunk {
469
618
  * Chunk returned by the sdk during streaming chat completions.
470
619
  */
471
620
  export type StreamChunk = ContentStreamChunk | ToolCallStreamChunk | ToolResultStreamChunk | DoneStreamChunk | ErrorStreamChunk | ApprovalRequestedStreamChunk | ToolInputAvailableStreamChunk | ThinkingStreamChunk;
472
- export interface ChatCompletionChunk {
621
+ export interface TextCompletionChunk {
473
622
  id: string;
474
623
  model: string;
475
624
  content: string;
@@ -498,20 +647,207 @@ export interface SummarizationResult {
498
647
  totalTokens: number;
499
648
  };
500
649
  }
501
- export interface EmbeddingOptions {
650
+ /**
651
+ * Options for image generation.
652
+ * These are the common options supported across providers.
653
+ */
654
+ export interface ImageGenerationOptions<TProviderOptions extends object = object> {
655
+ /** The model to use for image generation */
502
656
  model: string;
503
- input: string | Array<string>;
504
- dimensions?: number;
657
+ /** Text description of the desired image(s) */
658
+ prompt: string;
659
+ /** Number of images to generate (default: 1) */
660
+ numberOfImages?: number;
661
+ /** Image size in WIDTHxHEIGHT format (e.g., "1024x1024") */
662
+ size?: string;
663
+ /** Model-specific options for image generation */
664
+ modelOptions?: TProviderOptions;
505
665
  }
506
- export interface EmbeddingResult {
666
+ /**
667
+ * A single generated image
668
+ */
669
+ export interface GeneratedImage {
670
+ /** Base64-encoded image data */
671
+ b64Json?: string;
672
+ /** URL to the generated image (may be temporary) */
673
+ url?: string;
674
+ /** Revised prompt used by the model (if applicable) */
675
+ revisedPrompt?: string;
676
+ }
677
+ /**
678
+ * Result of image generation
679
+ */
680
+ export interface ImageGenerationResult {
681
+ /** Unique identifier for the generation */
507
682
  id: string;
683
+ /** Model used for generation */
508
684
  model: string;
509
- embeddings: Array<Array<number>>;
510
- usage: {
511
- promptTokens: number;
512
- totalTokens: number;
685
+ /** Array of generated images */
686
+ images: Array<GeneratedImage>;
687
+ /** Token usage information (if available) */
688
+ usage?: {
689
+ inputTokens?: number;
690
+ outputTokens?: number;
691
+ totalTokens?: number;
513
692
  };
514
693
  }
694
+ /**
695
+ * Options for video generation.
696
+ * These are the common options supported across providers.
697
+ *
698
+ * @experimental Video generation is an experimental feature and may change.
699
+ */
700
+ export interface VideoGenerationOptions<TProviderOptions extends object = object> {
701
+ /** The model to use for video generation */
702
+ model: string;
703
+ /** Text description of the desired video */
704
+ prompt: string;
705
+ /** Video size in WIDTHxHEIGHT format (e.g., "1280x720") */
706
+ size?: string;
707
+ /** Video duration in seconds */
708
+ duration?: number;
709
+ /** Model-specific options for video generation */
710
+ modelOptions?: TProviderOptions;
711
+ }
712
+ /**
713
+ * Result of creating a video generation job.
714
+ *
715
+ * @experimental Video generation is an experimental feature and may change.
716
+ */
717
+ export interface VideoJobResult {
718
+ /** Unique job identifier for polling status */
719
+ jobId: string;
720
+ /** Model used for generation */
721
+ model: string;
722
+ }
723
+ /**
724
+ * Status of a video generation job.
725
+ *
726
+ * @experimental Video generation is an experimental feature and may change.
727
+ */
728
+ export interface VideoStatusResult {
729
+ /** Job identifier */
730
+ jobId: string;
731
+ /** Current status of the job */
732
+ status: 'pending' | 'processing' | 'completed' | 'failed';
733
+ /** Progress percentage (0-100), if available */
734
+ progress?: number;
735
+ /** Error message if status is 'failed' */
736
+ error?: string;
737
+ }
738
+ /**
739
+ * Result containing the URL to a generated video.
740
+ *
741
+ * @experimental Video generation is an experimental feature and may change.
742
+ */
743
+ export interface VideoUrlResult {
744
+ /** Job identifier */
745
+ jobId: string;
746
+ /** URL to the generated video */
747
+ url: string;
748
+ /** When the URL expires, if applicable */
749
+ expiresAt?: Date;
750
+ }
751
+ /**
752
+ * Options for text-to-speech generation.
753
+ * These are the common options supported across providers.
754
+ */
755
+ export interface TTSOptions<TProviderOptions extends object = object> {
756
+ /** The model to use for TTS generation */
757
+ model: string;
758
+ /** The text to convert to speech */
759
+ text: string;
760
+ /** The voice to use for generation */
761
+ voice?: string;
762
+ /** The output audio format */
763
+ format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm';
764
+ /** The speed of the generated audio (0.25 to 4.0) */
765
+ speed?: number;
766
+ /** Model-specific options for TTS generation */
767
+ modelOptions?: TProviderOptions;
768
+ }
769
+ /**
770
+ * Result of text-to-speech generation.
771
+ */
772
+ export interface TTSResult {
773
+ /** Unique identifier for the generation */
774
+ id: string;
775
+ /** Model used for generation */
776
+ model: string;
777
+ /** Base64-encoded audio data */
778
+ audio: string;
779
+ /** Audio format of the generated audio */
780
+ format: string;
781
+ /** Duration of the audio in seconds, if available */
782
+ duration?: number;
783
+ /** Content type of the audio (e.g., 'audio/mp3') */
784
+ contentType?: string;
785
+ }
786
+ /**
787
+ * Options for audio transcription.
788
+ * These are the common options supported across providers.
789
+ */
790
+ export interface TranscriptionOptions<TProviderOptions extends object = object> {
791
+ /** The model to use for transcription */
792
+ model: string;
793
+ /** The audio data to transcribe - can be base64 string, File, Blob, or Buffer */
794
+ audio: string | File | Blob | ArrayBuffer;
795
+ /** The language of the audio in ISO-639-1 format (e.g., 'en') */
796
+ language?: string;
797
+ /** An optional prompt to guide the transcription */
798
+ prompt?: string;
799
+ /** The format of the transcription output */
800
+ responseFormat?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
801
+ /** Model-specific options for transcription */
802
+ modelOptions?: TProviderOptions;
803
+ }
804
+ /**
805
+ * A single segment of transcribed audio with timing information.
806
+ */
807
+ export interface TranscriptionSegment {
808
+ /** Unique identifier for the segment */
809
+ id: number;
810
+ /** Start time of the segment in seconds */
811
+ start: number;
812
+ /** End time of the segment in seconds */
813
+ end: number;
814
+ /** Transcribed text for this segment */
815
+ text: string;
816
+ /** Confidence score (0-1), if available */
817
+ confidence?: number;
818
+ /** Speaker identifier, if diarization is enabled */
819
+ speaker?: string;
820
+ }
821
+ /**
822
+ * A single word with timing information.
823
+ */
824
+ export interface TranscriptionWord {
825
+ /** The transcribed word */
826
+ word: string;
827
+ /** Start time in seconds */
828
+ start: number;
829
+ /** End time in seconds */
830
+ end: number;
831
+ }
832
+ /**
833
+ * Result of audio transcription.
834
+ */
835
+ export interface TranscriptionResult {
836
+ /** Unique identifier for the transcription */
837
+ id: string;
838
+ /** Model used for transcription */
839
+ model: string;
840
+ /** The full transcribed text */
841
+ text: string;
842
+ /** Language detected or specified */
843
+ language?: string;
844
+ /** Duration of the audio in seconds */
845
+ duration?: number;
846
+ /** Detailed segments with timing, if available */
847
+ segments?: Array<TranscriptionSegment>;
848
+ /** Word-level timestamps, if available */
849
+ words?: Array<TranscriptionWord>;
850
+ }
515
851
  /**
516
852
  * Default metadata type for adapters that don't define custom metadata.
517
853
  * Uses unknown for all modalities.
@@ -523,102 +859,3 @@ export interface DefaultMessageMetadataByModality {
523
859
  video: unknown;
524
860
  document: unknown;
525
861
  }
526
- /**
527
- * AI adapter interface with support for endpoint-specific models and provider options.
528
- *
529
- * Generic parameters:
530
- * - TChatModels: Models that support chat/text completion
531
- * - TEmbeddingModels: Models that support embeddings
532
- * - TChatProviderOptions: Provider-specific options for chat endpoint
533
- * - TEmbeddingProviderOptions: Provider-specific options for embedding endpoint
534
- * - TModelProviderOptionsByName: Map from model name to its specific provider options
535
- * - TModelInputModalitiesByName: Map from model name to its supported input modalities
536
- * - TMessageMetadataByModality: Map from modality type to adapter-specific metadata types
537
- */
538
- export interface AIAdapter<TChatModels extends ReadonlyArray<string> = ReadonlyArray<string>, TEmbeddingModels extends ReadonlyArray<string> = ReadonlyArray<string>, TChatProviderOptions extends Record<string, any> = Record<string, any>, TEmbeddingProviderOptions extends Record<string, any> = Record<string, any>, TModelProviderOptionsByName extends Record<string, any> = Record<string, any>, TModelInputModalitiesByName extends Record<string, ReadonlyArray<Modality>> = Record<string, ReadonlyArray<Modality>>, TMessageMetadataByModality extends {
539
- text: unknown;
540
- image: unknown;
541
- audio: unknown;
542
- video: unknown;
543
- document: unknown;
544
- } = DefaultMessageMetadataByModality> {
545
- name: string;
546
- /** Models that support chat/text completion */
547
- models: TChatModels;
548
- /** Models that support embeddings */
549
- embeddingModels?: TEmbeddingModels;
550
- _providerOptions?: TChatProviderOptions;
551
- _chatProviderOptions?: TChatProviderOptions;
552
- _embeddingProviderOptions?: TEmbeddingProviderOptions;
553
- /**
554
- * Type-only map from model name to its specific provider options.
555
- * Used by the core AI types to narrow providerOptions based on the selected model.
556
- * Must be provided by all adapters.
557
- */
558
- _modelProviderOptionsByName: TModelProviderOptionsByName;
559
- /**
560
- * Type-only map from model name to its supported input modalities.
561
- * Used by the core AI types to narrow ContentPart types based on the selected model.
562
- * Must be provided by all adapters.
563
- */
564
- _modelInputModalitiesByName?: TModelInputModalitiesByName;
565
- /**
566
- * Type-only map from modality type to adapter-specific metadata types.
567
- * Used to provide type-safe autocomplete for metadata on content parts.
568
- */
569
- _messageMetadataByModality?: TMessageMetadataByModality;
570
- chatStream: (options: ChatOptions<string, TChatProviderOptions>) => AsyncIterable<StreamChunk>;
571
- summarize: (options: SummarizationOptions) => Promise<SummarizationResult>;
572
- createEmbeddings: (options: EmbeddingOptions) => Promise<EmbeddingResult>;
573
- }
574
- export interface AIAdapterConfig {
575
- apiKey?: string;
576
- baseUrl?: string;
577
- timeout?: number;
578
- maxRetries?: number;
579
- headers?: Record<string, string>;
580
- }
581
- export type ChatStreamOptionsUnion<TAdapter extends AIAdapter<any, any, any, any, any, any, any>> = TAdapter extends AIAdapter<infer Models, any, any, any, infer ModelProviderOptions, infer ModelInputModalities, infer MessageMetadata> ? Models[number] extends infer TModel ? TModel extends string ? Omit<ChatOptions, 'model' | 'providerOptions' | 'responseFormat' | 'messages'> & {
582
- adapter: TAdapter;
583
- model: TModel;
584
- providerOptions?: TModel extends keyof ModelProviderOptions ? ModelProviderOptions[TModel] : never;
585
- /**
586
- * Messages array with content constrained to the model's supported input modalities.
587
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
588
- * Metadata types are also constrained based on the adapter's metadata type definitions.
589
- */
590
- messages: TModel extends keyof ModelInputModalities ? ModelInputModalities[TModel] extends ReadonlyArray<Modality> ? MessageMetadata extends {
591
- text: infer TTextMeta;
592
- image: infer TImageMeta;
593
- audio: infer TAudioMeta;
594
- video: infer TVideoMeta;
595
- document: infer TDocumentMeta;
596
- } ? Array<ConstrainedModelMessage<ModelInputModalities[TModel], TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>> : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>> : Array<ModelMessage> : Array<ModelMessage>;
597
- } : never : never : never;
598
- /**
599
- * Chat options constrained by a specific model's capabilities.
600
- * Unlike ChatStreamOptionsUnion which creates a union over all models,
601
- * this type takes a specific model and constrains messages accordingly.
602
- */
603
- export type ChatStreamOptionsForModel<TAdapter extends AIAdapter<any, any, any, any, any, any, any>, TModel extends string> = TAdapter extends AIAdapter<any, any, any, any, infer ModelProviderOptions, infer ModelInputModalities, infer MessageMetadata> ? Omit<ChatOptions, 'model' | 'providerOptions' | 'responseFormat' | 'messages'> & {
604
- adapter: TAdapter;
605
- model: TModel;
606
- providerOptions?: TModel extends keyof ModelProviderOptions ? ModelProviderOptions[TModel] : never;
607
- /**
608
- * Messages array with content constrained to the model's supported input modalities.
609
- * For example, if a model only supports ['text', 'image'], you cannot pass audio or video content.
610
- * Metadata types are also constrained based on the adapter's metadata type definitions.
611
- */
612
- messages: TModel extends keyof ModelInputModalities ? ModelInputModalities[TModel] extends ReadonlyArray<Modality> ? MessageMetadata extends {
613
- text: infer TTextMeta;
614
- image: infer TImageMeta;
615
- audio: infer TAudioMeta;
616
- video: infer TVideoMeta;
617
- document: infer TDocumentMeta;
618
- } ? Array<ConstrainedModelMessage<ModelInputModalities[TModel], TImageMeta, TAudioMeta, TVideoMeta, TDocumentMeta, TTextMeta>> : Array<ConstrainedModelMessage<ModelInputModalities[TModel]>> : Array<ModelMessage> : Array<ModelMessage>;
619
- } : never;
620
- export type ExtractModelsFromAdapter<T> = T extends AIAdapter<infer M, any, any, any, any, any> ? M[number] : never;
621
- /**
622
- * Extract the supported input modalities for a specific model from an adapter.
623
- */
624
- export type ExtractModalitiesForModel<TAdapter extends AIAdapter<any, any, any, any, any, any>, TModel extends string> = TAdapter extends AIAdapter<any, any, any, any, any, infer ModelInputModalities> ? TModel extends keyof ModelInputModalities ? ModelInputModalities[TModel] : ReadonlyArray<Modality> : ReadonlyArray<Modality>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.0.3",
3
+ "version": "0.2.0",
4
4
  "description": "Core TanStack AI library - Open source AI SDK",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -17,6 +17,10 @@
17
17
  "types": "./dist/esm/index.d.ts",
18
18
  "import": "./dist/esm/index.js"
19
19
  },
20
+ "./adapters": {
21
+ "types": "./dist/esm/activities/index.d.ts",
22
+ "import": "./dist/esm/activities/index.js"
23
+ },
20
24
  "./event-client": {
21
25
  "types": "./dist/esm/event-client.d.ts",
22
26
  "import": "./dist/esm/event-client.js"
@@ -42,13 +46,10 @@
42
46
  "@tanstack/devtools-event-client": "^0.4.0",
43
47
  "partial-json": "^0.1.7"
44
48
  },
45
- "peerDependencies": {
46
- "@alcyone-labs/zod-to-json-schema": "^4.0.0",
47
- "zod": "^3.0.0 || ^4.0.0"
48
- },
49
49
  "devDependencies": {
50
+ "@standard-schema/spec": "^1.1.0",
50
51
  "@vitest/coverage-v8": "4.0.14",
51
- "zod": "^4.1.13"
52
+ "zod": "^4.2.0"
52
53
  },
53
54
  "scripts": {
54
55
  "build": "vite build",