@tanstack/ai 0.39.1 → 0.41.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/dist/esm/activities/chat/index.d.ts +2 -2
  2. package/dist/esm/activities/chat/index.js +20 -3
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/messages.js +7 -0
  5. package/dist/esm/activities/chat/messages.js.map +1 -1
  6. package/dist/esm/activities/chat/middleware/compose.d.ts +2 -2
  7. package/dist/esm/activities/chat/middleware/compose.js.map +1 -1
  8. package/dist/esm/activities/chat/middleware/index.d.ts +1 -1
  9. package/dist/esm/activities/chat/middleware/sandbox-runtime.d.ts +7 -2
  10. package/dist/esm/activities/chat/middleware/sandbox-runtime.js.map +1 -1
  11. package/dist/esm/activities/chat/middleware/types.d.ts +16 -4
  12. package/dist/esm/activities/chat/stream/message-updaters.d.ts +2 -0
  13. package/dist/esm/activities/chat/stream/message-updaters.js +3 -1
  14. package/dist/esm/activities/chat/stream/message-updaters.js.map +1 -1
  15. package/dist/esm/activities/chat/stream/processor.js +18 -1
  16. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  17. package/dist/esm/activities/chat/tools/tool-definition.d.ts +14 -11
  18. package/dist/esm/activities/chat/tools/tool-definition.js.map +1 -1
  19. package/dist/esm/activities/generateAudio/index.d.ts +1 -1
  20. package/dist/esm/activities/generateAudio/index.js.map +1 -1
  21. package/dist/esm/activities/generateImage/index.d.ts +1 -1
  22. package/dist/esm/activities/generateImage/index.js +1 -1
  23. package/dist/esm/activities/generateImage/index.js.map +1 -1
  24. package/dist/esm/activities/generateSpeech/index.d.ts +1 -1
  25. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  26. package/dist/esm/activities/generateTranscription/index.d.ts +3 -3
  27. package/dist/esm/activities/generateTranscription/index.js.map +1 -1
  28. package/dist/esm/activities/generateVideo/index.d.ts +12 -12
  29. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  30. package/dist/esm/index.d.ts +3 -3
  31. package/dist/esm/index.js +2 -0
  32. package/dist/esm/index.js.map +1 -1
  33. package/dist/esm/realtime/event-emitter.d.ts +5 -0
  34. package/dist/esm/realtime/event-emitter.js +27 -0
  35. package/dist/esm/realtime/event-emitter.js.map +1 -0
  36. package/dist/esm/realtime/index.d.ts +1 -0
  37. package/dist/esm/realtime/index.js.map +1 -1
  38. package/dist/esm/realtime/types.d.ts +9 -1
  39. package/dist/esm/types.d.ts +124 -1
  40. package/package.json +1 -1
  41. package/skills/ai-core/adapter-configuration/SKILL.md +10 -0
  42. package/skills/ai-core/ag-ui-protocol/SKILL.md +59 -0
  43. package/skills/ai-core/media-generation/SKILL.md +50 -5
  44. package/skills/ai-core/middleware/SKILL.md +106 -0
  45. package/skills/ai-core/tool-calling/SKILL.md +13 -1
  46. package/src/activities/chat/index.ts +26 -6
  47. package/src/activities/chat/messages.ts +9 -0
  48. package/src/activities/chat/middleware/compose.ts +2 -2
  49. package/src/activities/chat/middleware/index.ts +1 -0
  50. package/src/activities/chat/middleware/sandbox-runtime.ts +4 -2
  51. package/src/activities/chat/middleware/types.ts +17 -4
  52. package/src/activities/chat/stream/message-updaters.ts +7 -1
  53. package/src/activities/chat/stream/processor.ts +30 -2
  54. package/src/activities/chat/tools/tool-definition.ts +33 -11
  55. package/src/activities/generateAudio/index.ts +2 -2
  56. package/src/activities/generateImage/index.ts +2 -2
  57. package/src/activities/generateSpeech/index.ts +2 -2
  58. package/src/activities/generateTranscription/index.ts +8 -4
  59. package/src/activities/generateVideo/index.ts +29 -15
  60. package/src/index.ts +3 -1
  61. package/src/realtime/event-emitter.ts +46 -0
  62. package/src/realtime/index.ts +2 -0
  63. package/src/realtime/types.ts +9 -0
  64. package/src/types.ts +116 -1
@@ -18,7 +18,7 @@ export type { ProviderTool } from './tools/provider-tool.js';
18
18
  export { brandProviderTool } from './tools/provider-tool.js';
19
19
  export { maxIterations, untilFinishReason, combineStrategies, } from './activities/chat/agent-loop-strategies.js';
20
20
  export { createToolRegistry, createFrozenRegistry, type ToolRegistry, } from './tool-registry.js';
21
- export type { ChatMiddleware, ChatMiddlewareContext, ChatMiddlewarePhase, ChatMiddlewareConfig, StructuredOutputMiddlewareConfig, ToolCallHookContext, BeforeToolCallDecision, AfterToolCallInfo, IterationInfo, ToolPhaseCompleteInfo, UsageInfo, FinishInfo, AbortInfo, ErrorInfo, SandboxFileEvent, ChatSandboxHooks, } from './activities/chat/middleware/index.js';
21
+ export type { ChatMiddleware, ChatMiddlewareContext, ChatMiddlewarePhase, ChatMiddlewareConfig, StructuredOutputMiddlewareConfig, ToolCallHookContext, BeforeToolCallDecision, AfterToolCallInfo, IterationInfo, ToolPhaseCompleteInfo, UsageInfo, FinishInfo, AbortInfo, ErrorInfo, SandboxFileEvent, SandboxFileHookEvent, ChatSandboxHooks, } from './activities/chat/middleware/index.js';
22
22
  export type { GenerationMiddleware, GenerationMiddlewareContext, GenerationActivity, GenerationUsageInfo, GenerationFinishInfo, GenerationAbortInfo, GenerationErrorInfo, AnyGenerationMiddleware, } from './activities/middleware/index.js';
23
23
  export { createCapability, defineChatMiddleware, createChatMiddleware, } from './activities/chat/middleware/index.js';
24
24
  export type { Capability, CapabilityHandle, CapabilityContext, CapabilityGetter, CapabilityProvider, DefinedChatMiddleware, AnyChatMiddleware, } from './activities/chat/middleware/index.js';
@@ -30,8 +30,8 @@ export type { ResolvedMediaPrompt } from './utilities/media-prompt.js';
30
30
  export type { SystemPrompt, NormalizedSystemPrompt } from './system-prompts.js';
31
31
  export { normalizeSystemPrompts } from './system-prompts.js';
32
32
  export { detectImageMimeType } from './utils.js';
33
- export { realtimeToken } from './realtime/index.js';
34
- export type { RealtimeToken, RealtimeTokenAdapter, RealtimeTokenOptions, RealtimeSessionConfig, VADConfig, RealtimeMessage, RealtimeMessagePart, RealtimeTextPart, RealtimeAudioPart, RealtimeToolCallPart, RealtimeToolResultPart, RealtimeImagePart, RealtimeStatus, RealtimeMode, AudioVisualization, RealtimeEvent, RealtimeEventPayloads, RealtimeEventHandler, RealtimeErrorCode, RealtimeError, RealtimeAdapter, RealtimeConnection, } from './realtime/index.js';
33
+ export { realtimeToken, createRealtimeEventEmitter } from './realtime/index.js';
34
+ export type { RealtimeToken, RealtimeTokenAdapter, RealtimeTokenOptions, RealtimeSessionConfig, RealtimeToolConfig, VADConfig, RealtimeMessage, RealtimeMessagePart, RealtimeTextPart, RealtimeAudioPart, RealtimeToolCallPart, RealtimeToolResultPart, RealtimeImagePart, RealtimeStatus, RealtimeMode, AudioVisualization, RealtimeEvent, RealtimeEventPayloads, RealtimeEventHandler, RealtimeErrorCode, RealtimeError, RealtimeAdapter, RealtimeConnection, } from './realtime/index.js';
35
35
  export { convertMessagesToModelMessages, generateMessageId, uiMessageToModelMessages, modelMessageToUIMessage, modelMessagesToUIMessages, normalizeToUIMessage, } from './activities/chat/messages.js';
36
36
  export { StreamProcessor, createReplayStream, ImmediateStrategy, PunctuationStrategy, BatchStrategy, WordBoundaryStrategy, CompositeStrategy, PartialJSONParser, defaultJSONParser, parsePartialJSON, } from './activities/chat/stream/index.js';
37
37
  export type { ChunkStrategy, ChunkRecording, InternalToolCallState, ProcessorResult, ProcessorState, StreamProcessorEvents, StreamProcessorOptions, ToolCallState, ToolResultState, JSONParser, } from './activities/chat/stream/index.js';
package/dist/esm/index.js CHANGED
@@ -33,6 +33,7 @@ import { PartialJSONParser, defaultJSONParser, parsePartialJSON } from "./activi
33
33
  import { StreamProcessor, createReplayStream } from "./activities/chat/stream/processor.js";
34
34
  import { createCapability } from "./activities/chat/middleware/capabilities.js";
35
35
  import { createChatMiddleware } from "./activities/chat/middleware/builder.js";
36
+ import { createRealtimeEventEmitter } from "./realtime/event-emitter.js";
36
37
  import { defineChatMiddleware } from "./activities/chat/middleware/define.js";
37
38
  export {
38
39
  BatchStrategy,
@@ -63,6 +64,7 @@ export {
63
64
  createFrozenRegistry,
64
65
  createImageOptions,
65
66
  createModel,
67
+ createRealtimeEventEmitter,
66
68
  createReplayStream,
67
69
  createSpeechOptions,
68
70
  createSummarizeOptions,
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
1
+ {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
@@ -0,0 +1,5 @@
1
+ import { RealtimeEvent, RealtimeEventHandler, RealtimeEventPayloads } from './types.js';
2
+ export declare function createRealtimeEventEmitter(): {
3
+ emit<TEvent extends RealtimeEvent>(event: TEvent, payload: RealtimeEventPayloads[TEvent]): void;
4
+ on<TEvent extends RealtimeEvent>(event: TEvent, handler: RealtimeEventHandler<TEvent>): () => void;
5
+ };
@@ -0,0 +1,27 @@
1
+ function createRealtimeEventEmitter() {
2
+ const eventHandlers = /* @__PURE__ */ new Map();
3
+ return {
4
+ emit(event, payload) {
5
+ const handlers = eventHandlers.get(event);
6
+ if (!handlers) return;
7
+ for (const handler of handlers) {
8
+ handler(payload);
9
+ }
10
+ },
11
+ on(event, handler) {
12
+ let handlers = eventHandlers.get(event);
13
+ if (!handlers) {
14
+ handlers = /* @__PURE__ */ new Set();
15
+ eventHandlers.set(event, handlers);
16
+ }
17
+ handlers.add(handler);
18
+ return () => {
19
+ handlers.delete(handler);
20
+ };
21
+ }
22
+ };
23
+ }
24
+ export {
25
+ createRealtimeEventEmitter
26
+ };
27
+ //# sourceMappingURL=event-emitter.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"event-emitter.js","sources":["../../../src/realtime/event-emitter.ts"],"sourcesContent":["import type {\n RealtimeEvent,\n RealtimeEventHandler,\n RealtimeEventPayloads,\n} from './types'\n\n/**\n * Handlers are stored with a `never` payload so any specific\n * `RealtimeEventHandler<TEvent>` is assignable in (contravariance), keeping the\n * heterogeneous handler map type-safe without `any`. `emit` narrows back to the\n * event's real payload type via its signature; the lone `as never` at the call\n * site is the inverse of that stored `never`.\n */\ntype StoredHandler = (payload: never) => void\n\nexport function createRealtimeEventEmitter() {\n const eventHandlers = new Map<RealtimeEvent, Set<StoredHandler>>()\n\n return {\n emit<TEvent extends RealtimeEvent>(\n event: TEvent,\n payload: RealtimeEventPayloads[TEvent],\n ) {\n const handlers = eventHandlers.get(event)\n if (!handlers) return\n for (const handler of handlers) {\n handler(payload as never)\n }\n },\n on<TEvent extends RealtimeEvent>(\n event: TEvent,\n handler: RealtimeEventHandler<TEvent>,\n ): () => void {\n let handlers = eventHandlers.get(event)\n if (!handlers) {\n handlers = new Set<StoredHandler>()\n eventHandlers.set(event, handlers)\n }\n handlers.add(handler)\n\n return () => {\n handlers.delete(handler)\n }\n },\n }\n}\n"],"names":[],"mappings":"AAeO,SAAS,6BAA6B;AAC3C,QAAM,oCAAoB,IAAA;AAE1B,SAAO;AAAA,IACL,KACE,OACA,SACA;AACA,YAAM,WAAW,cAAc,IAAI,KAAK;AACxC,UAAI,CAAC,SAAU;AACf,iBAAW,WAAW,UAAU;AAC9B,gBAAQ,OAAgB;AAAA,MAC1B;AAAA,IACF;AAAA,IACA,GACE,OACA,SACY;AACZ,UAAI,WAAW,cAAc,IAAI,KAAK;AACtC,UAAI,CAAC,UAAU;AACb,uCAAe,IAAA;AACf,sBAAc,IAAI,OAAO,QAAQ;AAAA,MACnC;AACA,eAAS,IAAI,OAAO;AAEpB,aAAO,MAAM;AACX,iBAAS,OAAO,OAAO;AAAA,MACzB;AAAA,IACF;AAAA,EAAA;AAEJ;"}
@@ -1,4 +1,5 @@
1
1
  import { RealtimeToken, RealtimeTokenOptions } from './types.js';
2
+ export { createRealtimeEventEmitter } from './event-emitter.js';
2
3
  export type * from './types.js';
3
4
  /**
4
5
  * Generate a realtime token using the provided adapter.
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":["../../../src/realtime/index.ts"],"sourcesContent":["import type { RealtimeToken, RealtimeTokenOptions } from './types'\n\n// Re-export all types\nexport type * from './types'\n\n/**\n * Generate a realtime token using the provided adapter.\n *\n * This function is used on the server to generate ephemeral tokens\n * that clients can use to establish realtime connections.\n *\n * @param options - Token generation options including the adapter\n * @returns Promise resolving to a RealtimeToken\n *\n * @example\n * ```typescript\n * import { realtimeToken } from '@tanstack/ai'\n * import { openaiRealtimeToken } from '@tanstack/ai-openai'\n *\n * // Server function (TanStack Start example)\n * export const getRealtimeToken = createServerFn()\n * .handler(async () => {\n * return realtimeToken({\n * adapter: openaiRealtimeToken({\n * model: 'gpt-realtime',\n * }),\n * })\n * })\n * ```\n */\nexport async function realtimeToken(\n options: RealtimeTokenOptions,\n): Promise<RealtimeToken> {\n const { adapter } = options\n return adapter.generateToken()\n}\n"],"names":[],"mappings":"AA8BA,eAAsB,cACpB,SACwB;AACxB,QAAM,EAAE,YAAY;AACpB,SAAO,QAAQ,cAAA;AACjB;"}
1
+ {"version":3,"file":"index.js","sources":["../../../src/realtime/index.ts"],"sourcesContent":["import type { RealtimeToken, RealtimeTokenOptions } from './types'\n\nexport { createRealtimeEventEmitter } from './event-emitter'\n\n// Re-export all types\nexport type * from './types'\n\n/**\n * Generate a realtime token using the provided adapter.\n *\n * This function is used on the server to generate ephemeral tokens\n * that clients can use to establish realtime connections.\n *\n * @param options - Token generation options including the adapter\n * @returns Promise resolving to a RealtimeToken\n *\n * @example\n * ```typescript\n * import { realtimeToken } from '@tanstack/ai'\n * import { openaiRealtimeToken } from '@tanstack/ai-openai'\n *\n * // Server function (TanStack Start example)\n * export const getRealtimeToken = createServerFn()\n * .handler(async () => {\n * return realtimeToken({\n * adapter: openaiRealtimeToken({\n * model: 'gpt-realtime',\n * }),\n * })\n * })\n * ```\n */\nexport async function realtimeToken(\n options: RealtimeTokenOptions,\n): Promise<RealtimeToken> {\n const { adapter } = options\n return adapter.generateToken()\n}\n"],"names":[],"mappings":"AAgCA,eAAsB,cACpB,SACwB;AACxB,QAAM,EAAE,YAAY;AACpB,SAAO,QAAQ,cAAA;AACjB;"}
@@ -1,4 +1,5 @@
1
1
  import { AnyClientTool } from '../activities/chat/tools/tool-definition.js';
2
+ import { UsageInfo } from '../activities/chat/middleware/types.js';
2
3
  /**
3
4
  * Voice activity detection configuration
4
5
  */
@@ -18,6 +19,7 @@ export interface RealtimeToolConfig {
18
19
  name: string;
19
20
  description: string;
20
21
  inputSchema?: Record<string, any>;
22
+ outputSchema?: Record<string, any>;
21
23
  }
22
24
  /**
23
25
  * Configuration for a realtime session
@@ -182,7 +184,7 @@ export interface AudioVisualization {
182
184
  /**
183
185
  * Events emitted by the realtime connection
184
186
  */
185
- export type RealtimeEvent = 'status_change' | 'mode_change' | 'transcript' | 'audio_chunk' | 'tool_call' | 'message_complete' | 'interrupted' | 'error';
187
+ export type RealtimeEvent = 'status_change' | 'mode_change' | 'transcript' | 'audio_chunk' | 'tool_call' | 'message_complete' | 'interrupted' | 'error' | 'go_away' | 'usage';
186
188
  /**
187
189
  * Event payloads for realtime events
188
190
  */
@@ -216,6 +218,10 @@ export interface RealtimeEventPayloads {
216
218
  error: {
217
219
  error: Error;
218
220
  };
221
+ go_away: {
222
+ timeLeft?: string;
223
+ };
224
+ usage: UsageInfo;
219
225
  }
220
226
  /**
221
227
  * Handler type for realtime events
@@ -273,6 +279,8 @@ export interface RealtimeConnection {
273
279
  sendToolResult: (callId: string, result: string) => void;
274
280
  /** Update session configuration */
275
281
  updateSession: (config: Partial<RealtimeSessionConfig>) => void;
282
+ /** Update the ephemeral token (e.g. on refresh); provider may reconnect */
283
+ updateToken?: (token: RealtimeToken) => void;
276
284
  /** Interrupt the current response */
277
285
  interrupt: () => void;
278
286
  /** Subscribe to connection events */
@@ -261,6 +261,15 @@ export interface ToolCallPart<TMetadata = unknown> {
261
261
  id: string;
262
262
  name: string;
263
263
  arguments: string;
264
+ /**
265
+ * Parsed tool input. Set from the parsed arguments once they are complete
266
+ * (`state: 'input-complete'` and later). `undefined` while the raw
267
+ * `arguments` string is still streaming, and may stay `undefined` for a call
268
+ * that terminates in an error state — the raw `arguments` string is always
269
+ * available as a fallback. Typed per-tool on the client `ToolCallPart` (see
270
+ * `@tanstack/ai-client`); `unknown` on this base type.
271
+ */
272
+ input?: unknown;
264
273
  state: ToolCallState;
265
274
  /** Approval metadata if tool requires user approval */
266
275
  approval?: {
@@ -1154,6 +1163,119 @@ export interface UIResourceEvent extends CustomEvent {
1154
1163
  meta?: Record<string, unknown>;
1155
1164
  };
1156
1165
  }
1166
+ export interface SandboxFileCustomEvent extends CustomEvent {
1167
+ name: 'sandbox.file';
1168
+ value: {
1169
+ type: 'create' | 'change' | 'delete';
1170
+ path: string;
1171
+ timestamp: number;
1172
+ };
1173
+ }
1174
+ export interface SandboxFileDiffEvent extends CustomEvent {
1175
+ name: 'sandbox.file.diff';
1176
+ value: {
1177
+ path: string;
1178
+ diff: string;
1179
+ };
1180
+ }
1181
+ export interface FileChangedEvent extends CustomEvent {
1182
+ name: 'file.changed';
1183
+ value: {
1184
+ path: string;
1185
+ diff: string;
1186
+ };
1187
+ }
1188
+ export interface SessionIdEvent extends CustomEvent {
1189
+ name: `${string}.session-id`;
1190
+ value: {
1191
+ sessionId: string;
1192
+ };
1193
+ }
1194
+ export interface CodeModeExecutionStartedEvent extends CustomEvent {
1195
+ name: 'code_mode:execution_started';
1196
+ value: {
1197
+ timestamp: number;
1198
+ codeLength: number;
1199
+ };
1200
+ }
1201
+ export interface CodeModeConsoleEvent extends CustomEvent {
1202
+ name: 'code_mode:console';
1203
+ value: {
1204
+ level: 'log' | 'warn' | 'error' | 'info';
1205
+ message: string;
1206
+ timestamp: number;
1207
+ };
1208
+ }
1209
+ export interface CodeModeExternalCallEvent extends CustomEvent {
1210
+ name: 'code_mode:external_call';
1211
+ value: {
1212
+ function: string;
1213
+ args: unknown;
1214
+ timestamp: number;
1215
+ };
1216
+ }
1217
+ export interface CodeModeExternalResultEvent extends CustomEvent {
1218
+ name: 'code_mode:external_result';
1219
+ value: {
1220
+ function: string;
1221
+ result: unknown;
1222
+ duration: number;
1223
+ };
1224
+ }
1225
+ export interface CodeModeExternalErrorEvent extends CustomEvent {
1226
+ name: 'code_mode:external_error';
1227
+ value: {
1228
+ function: string;
1229
+ error: string;
1230
+ duration: number;
1231
+ };
1232
+ }
1233
+ export interface CodeModeSkillCallEvent extends CustomEvent {
1234
+ name: 'code_mode:skill_call';
1235
+ value: {
1236
+ skill: string;
1237
+ input: unknown;
1238
+ timestamp: number;
1239
+ };
1240
+ }
1241
+ export interface CodeModeSkillResultEvent extends CustomEvent {
1242
+ name: 'code_mode:skill_result';
1243
+ value: {
1244
+ skill: string;
1245
+ result: unknown;
1246
+ duration: number;
1247
+ timestamp: number;
1248
+ };
1249
+ }
1250
+ export interface CodeModeSkillErrorEvent extends CustomEvent {
1251
+ name: 'code_mode:skill_error';
1252
+ value: {
1253
+ skill: string;
1254
+ error: string;
1255
+ duration: number;
1256
+ timestamp: number;
1257
+ };
1258
+ }
1259
+ export interface SkillRegisteredEvent extends CustomEvent {
1260
+ name: 'skill:registered';
1261
+ value: {
1262
+ id: string;
1263
+ name: string;
1264
+ description: string;
1265
+ timestamp: number;
1266
+ };
1267
+ }
1268
+ /**
1269
+ * Every CUSTOM event TanStack AI itself emits, as a discriminated union on
1270
+ * `name`. User-emitted custom events (via `emitCustomEvent` with a custom name)
1271
+ * are intentionally absent — they still flow at runtime.
1272
+ */
1273
+ export type KnownCustomEvent = SandboxFileCustomEvent | SandboxFileDiffEvent | FileChangedEvent | SessionIdEvent | CodeModeExecutionStartedEvent | CodeModeConsoleEvent | CodeModeExternalCallEvent | CodeModeExternalResultEvent | CodeModeExternalErrorEvent | CodeModeSkillCallEvent | CodeModeSkillResultEvent | CodeModeSkillErrorEvent | SkillRegisteredEvent | StructuredOutputStartEvent | StructuredOutputCompleteEvent | ApprovalRequestedEvent | ToolInputAvailableEvent | UIResourceEvent;
1274
+ /** The default chat streaming result: standard chunks plus every typed
1275
+ * framework CUSTOM event, with the `value: any` catch-all excluded so
1276
+ * literal-`name` narrowing types `value`. User-emitted custom names are typed
1277
+ * out (still flow at runtime — branch outside the name narrows or cast). */
1278
+ export type ChatStream = AsyncIterable<Exclude<StreamChunk, CustomEvent> | KnownCustomEvent>;
1157
1279
  /**
1158
1280
  * Public type for streams returned by `chat({ outputSchema, stream: true })`.
1159
1281
  *
@@ -1584,6 +1706,7 @@ export interface TTSResult {
1584
1706
  * Options for audio transcription.
1585
1707
  * These are the common options supported across providers.
1586
1708
  */
1709
+ export type TranscriptionResponseFormat = 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
1587
1710
  export interface TranscriptionOptions<TProviderOptions extends object = object> {
1588
1711
  /** The model to use for transcription */
1589
1712
  model: string;
@@ -1594,7 +1717,7 @@ export interface TranscriptionOptions<TProviderOptions extends object = object>
1594
1717
  /** An optional prompt to guide the transcription */
1595
1718
  prompt?: string;
1596
1719
  /** The format of the transcription output */
1597
- responseFormat?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
1720
+ responseFormat?: TranscriptionResponseFormat;
1598
1721
  /** Model-specific options for transcription */
1599
1722
  modelOptions?: TProviderOptions;
1600
1723
  /**
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.39.1",
3
+ "version": "0.41.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -297,6 +297,16 @@ Per-provider sampling keys (all live inside `modelOptions`):
297
297
  some sampling options use provider-native names. Ollama nests all sampling under
298
298
  `modelOptions.options`.
299
299
 
300
+ > **Anthropic `max_tokens` default:** Anthropic's API _requires_ `max_tokens`,
301
+ > so the adapter always sends one. When you omit `modelOptions.max_tokens`, it
302
+ > defaults to the selected model's full output ceiling (its `max_output_tokens`
303
+ > from model metadata — e.g. 64K for Sonnet, 128K for Opus), not a low constant.
304
+ > `max_tokens` is a ceiling, not a reservation (billing is per token generated),
305
+ > so leaving it unset is the right default for codegen / agentic / long-form
306
+ > output and avoids silent `stop_reason: "max_tokens"` truncation. Set it only to
307
+ > cap output below the model ceiling. Other providers treat token limits as
308
+ > optional and don't apply this flooring.
309
+
300
310
  ### 6. Capability Flag: `supportsCombinedToolsAndSchema`
301
311
 
302
312
  Adapters can declare an optional capability method:
@@ -13,6 +13,7 @@ sources:
13
13
  - 'TanStack/ai:docs/protocol/chunk-definitions.md'
14
14
  - 'TanStack/ai:docs/protocol/sse-protocol.md'
15
15
  - 'TanStack/ai:docs/protocol/http-stream-protocol.md'
16
+ - 'TanStack/ai:docs/protocol/custom-events.md'
16
17
  ---
17
18
 
18
19
  # AG-UI Protocol
@@ -218,6 +219,62 @@ RUN_STARTED -> TEXT_MESSAGE_START -> TEXT_MESSAGE_CONTENT* -> TEXT_MESSAGE_END
218
219
  union of all event interfaces). `StreamChunkType` is an alias for `AGUIEventType`
219
220
  (the string union of all event type literals).
220
221
 
222
+ ### 4. Typed CUSTOM Events — `ChatStream` and `KnownCustomEvent`
223
+
224
+ The `CUSTOM` row above describes the raw `StreamChunk` union, where the single
225
+ generic `CustomEvent` member types `value` as `any` -- once merged into a
226
+ union, that `any` poisons every other member too, so narrowing on `name`
227
+ still leaves `value: any`. `chat()` doesn't return raw `StreamChunk`; by
228
+ default (no `outputSchema`, `stream` not explicitly `false`) it returns
229
+ `ChatStream`, which swaps that generic member for `KnownCustomEvent` -- a
230
+ discriminated union of every `CUSTOM` event TanStack AI itself emits, each
231
+ with a literal `name` and a concrete `value`. Narrow with a plain `if` --
232
+ no helper, no cast:
233
+
234
+ ```typescript
235
+ import { chat } from '@tanstack/ai'
236
+ import { openaiText } from '@tanstack/ai-openai'
237
+
238
+ const stream = chat({
239
+ adapter: openaiText('gpt-5.2'),
240
+ messages,
241
+ })
242
+
243
+ for await (const chunk of stream) {
244
+ if (chunk.type === 'CUSTOM' && chunk.name === 'sandbox.file.diff') {
245
+ console.log(chunk.value.path, chunk.value.diff) // typed, no helper, no cast
246
+ } else if (
247
+ chunk.type === 'CUSTOM' &&
248
+ chunk.name === 'structured-output.complete'
249
+ ) {
250
+ console.log(chunk.value.object) // typed, no helper, no cast
251
+ }
252
+ }
253
+ ```
254
+
255
+ **Caveat -- `.endsWith()` (or any non-literal check) does not narrow.**
256
+ `SessionIdEvent['name']` is the template-literal type
257
+ `` `${string}.session-id` ``. TypeScript's control-flow narrowing only
258
+ understands exact comparisons (`===`) and `in`/type-predicate checks against
259
+ a discriminant -- a runtime `chunk.name.endsWith('.session-id')` check
260
+ doesn't inform the type system, so `chunk.value` stays the union of every
261
+ `KnownCustomEvent`'s `value`, not `{ sessionId: string }`. Compare against
262
+ the exact literal you expect, or write a user-defined type predicate
263
+ (`(c): c is SessionIdEvent => c.name.endsWith('.session-id')`) and call that
264
+ in the `if` instead.
265
+
266
+ **User-emitted `emitCustomEvent` names are typed out of `ChatStream`.** Tools
267
+ that call `context.emitCustomEvent('my-app:progress', ...)` still stream a
268
+ `CUSTOM` chunk at runtime, but `'my-app:progress'` isn't one of
269
+ `KnownCustomEvent`'s literal names, so it's intentionally absent from
270
+ `ChatStream`'s type -- including a generic fallback member would reintroduce
271
+ the `value: any` poison for every other event on the stream. To read your own
272
+ event with a type, annotate the stream as the wider `StreamChunk` instead of
273
+ `ChatStream` for that branch; its generic `CUSTOM` member already types
274
+ `value` as `any`, so no cast is needed there either.
275
+
276
+ Source: docs/protocol/custom-events.md
277
+
221
278
  ## Common Mistakes
222
279
 
223
280
  ### MEDIUM: Proxy buffering breaks SSE streaming
@@ -273,3 +330,5 @@ without transformation. See `docs/migration/ag-ui-compliance.md` for details.
273
330
  ## Cross-References
274
331
 
275
332
  - See also: `ai-core/custom-backend-integration/SKILL.md` -- Custom backends must implement SSE or HTTP stream format to work with TanStack AI client connection adapters.
333
+ - See also: `ai-core/middleware/SKILL.md` -- `sandbox.file.diff`'s `{ path, diff }` value (one of `KnownCustomEvent`'s members) is populated from the same lazy `before()`/`after()`/`diff()` accessors documented there for `onFile*` middleware hooks.
334
+ - Full CUSTOM event taxonomy: `docs/protocol/custom-events.md`.
@@ -261,6 +261,17 @@ await generateVideo({
261
261
  })
262
262
  ```
263
263
 
264
+ **URL inputs that require an upload throw by default.** Most adapters pass a
265
+ `type: 'url'` source straight through to the provider. Three paths can't —
266
+ OpenAI `images.edit()`, OpenAI Sora `input_reference`, and Gemini **Veo** —
267
+ because the provider only accepts uploaded bytes (Veo also takes a `gs://`
268
+ reference). For those, an HTTP(S) URL would have to be downloaded and buffered
269
+ in memory, which can OOM constrained runtimes, so they **throw** on an HTTP(S)
270
+ URL image input by default. Pass a `data:` URI (or `gs://` for Veo), or opt in
271
+ with `allowUrlFetch: true` on the adapter config
272
+ (`createOpenaiImage(model, apiKey, { allowUrlFetch: true })`, and likewise on
273
+ `createOpenaiVideo` / `createGeminiVideo`). `data:` URIs never need the flag.
274
+
264
275
  **Role hints** (`metadata.role`):
265
276
 
266
277
  | Role | Maps to |
@@ -357,7 +368,7 @@ const { generate, result, isLoading } = useGenerateSpeech({
357
368
  ### 4. Audio Transcription
358
369
 
359
370
  Adapter: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
360
- gpt-4o-mini-transcribe).
371
+ gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize).
361
372
 
362
373
  > **Capturing audio in the browser:** Use `useAudioRecorder` from `@tanstack/ai-react` to record directly in the browser, then pass the recording as the `audio` input to `generate()`, or use `recording.part` as a prompt part in chat/generation calls. No transcoding or extra dependencies required — the recorder returns the native browser format (`audio/webm` or `audio/mp4`). For transcription, wrap it as a `data:` URL so the provider gets the real content type; passing raw `recording.base64` makes the adapter assume `audio/mpeg` and mislabel the webm/mp4 bytes.
363
374
  >
@@ -382,16 +393,21 @@ const result = await generateTranscription({
382
393
  language: 'en',
383
394
  responseFormat: 'verbose_json',
384
395
  modelOptions: {
385
- include: ['segment', 'word'],
396
+ timestamp_granularities: ['word', 'segment'],
386
397
  },
387
398
  })
388
399
 
389
400
  // result.text -- full transcribed text
390
401
  // result.language -- detected/specified language
391
402
  // result.duration -- audio duration in seconds
392
- // result.segments -- timestamped segments with optional word-level timestamps
403
+ // result.segments -- timestamped segments (word-level timestamps are in result.words)
393
404
  ```
394
405
 
406
+ For speaker diarization, use `openaiTranscription('gpt-4o-transcribe-diarize')`.
407
+ When no response format is given it defaults the request to `response_format: 'diarized_json'`
408
+ and `chunking_strategy: 'auto'` (a top-level `responseFormat` of `'json'`/`'text'` opts out of
409
+ speaker segments); do not pass `prompt`, `include`, or `timestamp_granularities` with this model.
410
+
395
411
  Client hook:
396
412
 
397
413
  ```tsx
@@ -443,8 +459,8 @@ return toServerSentEventsResponse(stream)
443
459
  ```
444
460
 
445
461
  Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
446
- `duration` option is typed per model (e.g. `4 | 6 | 8` for Veo 3.x,
447
- `5 | 6 | 8` for Veo 2); use `adapter.snapDuration(seconds)` to coerce raw
462
+ `duration` option is typed per model (`4 | 6 | 8` for the Veo 3.1 models);
463
+ use `adapter.snapDuration(seconds)` to coerce raw
448
464
  seconds and `adapter.availableDurations()` to enumerate the valid set.
449
465
  Image prompt parts route by `metadata.role`: first un-roled /
450
466
  `'start_frame'` image → input image, `'end_frame'` → `lastFrame`,
@@ -467,6 +483,35 @@ const { jobId } = await generateVideo({
467
483
  // (x-goog-api-key header or ?key= query parameter).
468
484
  ```
469
485
 
486
+ Gemini Omni Flash (`geminiVideo('gemini-omni-flash-preview')`) is served by
487
+ the Interactions API instead of Veo's operations flow — same adapter, routed
488
+ by model. Clips are 720p; `duration` is any number of seconds in the 3–10
489
+ range (fractional ok, default 10 — availableDurations() reports the range),
490
+ `size` is the aspect ratio (`'16:9' | '9:16'`), and the finished video arrives
491
+ **inline** as a `data:video/mp4;base64,…` URL (no key needed to use it).
492
+ Image/video prompt parts are sent as interaction content blocks, grouped
493
+ as images, then videos, then text (no
494
+ `metadata.role` routing); `data` sources go inline, `url` sources pass
495
+ through as-is (never downloaded — use Gemini Files API URIs for remote
496
+ media). For conversational editing, pass a prior generation's `jobId` as
497
+ `modelOptions.previous_interaction_id` with a prompt describing the change:
498
+
499
+ ```typescript
500
+ import { geminiVideo } from '@tanstack/ai-gemini'
501
+
502
+ const omni = geminiVideo('gemini-omni-flash-preview')
503
+ const first = await generateVideo({
504
+ adapter: omni,
505
+ prompt: 'A violinist outdoors',
506
+ })
507
+ // …poll first.jobId to completion, then edit it:
508
+ const edited = await generateVideo({
509
+ adapter: omni,
510
+ prompt: 'Make the violin invisible',
511
+ modelOptions: { previous_interaction_id: first.jobId },
512
+ })
513
+ ```
514
+
470
515
  Other video adapters: `openaiVideo('sora-2')` (pixel sizes like `'1280x720'`,
471
516
  durations 4/8/12s, single `input_reference` image prompt part), `grokVideo(...)`
472
517
  (`grok-imagine-video` does text-to-video + image-to-video; `grok-imagine-video-1.5` is
@@ -11,6 +11,7 @@ library: tanstack-ai
11
11
  library_version: '0.10.0'
12
12
  sources:
13
13
  - 'TanStack/ai:docs/advanced/middleware.md'
14
+ - 'TanStack/ai:docs/sandbox/observability.md'
14
15
  ---
15
16
 
16
17
  # Middleware
@@ -371,6 +372,110 @@ Options: `maxSize` (default 100), `ttl` (default Infinity), `toolNames` (default
371
372
  `keyFn` (custom cache key), `storage` (custom backend like Redis). See
372
373
  `docs/advanced/middleware.md` for custom storage examples.
373
374
 
375
+ ## Sandbox File-Event Hooks (`sandbox` group)
376
+
377
+ Declare a `sandbox: ChatSandboxHooks` group on `defineChatMiddleware` to react
378
+ to every file created/changed/deleted inside a sandbox provided by
379
+ `withSandbox` (from `@tanstack/ai-sandbox`). These fire **per-run**,
380
+ server-side, and each handler receives the run's `ChatMiddlewareContext` as
381
+ the first argument:
382
+
383
+ ```typescript
384
+ import { defineChatMiddleware } from '@tanstack/ai'
385
+ import { db } from './db'
386
+
387
+ const auditMiddleware = defineChatMiddleware({
388
+ name: 'audit',
389
+ sandbox: {
390
+ onFile: (ctx, e) => console.log(ctx.runId, e.type, e.path),
391
+ onFileCreate: (ctx, e) => db.log({ run: ctx.runId, event: e }),
392
+ },
393
+ })
394
+ ```
395
+
396
+ | Hook | Fires for |
397
+ | -------------- | -------------------------- |
398
+ | `onFile` | Every create/change/delete |
399
+ | `onFileCreate` | File creates only |
400
+ | `onFileChange` | File changes only |
401
+ | `onFileDelete` | File deletes only |
402
+
403
+ These are independent of the stream: the engine also emits a `sandbox.file`
404
+ `CUSTOM` chunk per change regardless of whether any `sandbox` hooks are
405
+ registered, so a client can react to the same edits without middleware. See
406
+ `ai-core/ag-ui-protocol/SKILL.md` for reading that chunk (and the opt-in
407
+ `sandbox.file.diff` chunk) off `ChatStream`.
408
+
409
+ ### `before()` / `after()` / `diff()` — lazy, git-backed content accessors
410
+
411
+ Each hook receives a `SandboxFileHookEvent`: the serializable
412
+ `{ type, path, timestamp }` plus three lazy accessors for the file's content:
413
+
414
+ ```ts
415
+ interface SandboxFileHookEvent {
416
+ type: 'create' | 'change' | 'delete'
417
+ path: string
418
+ timestamp: number
419
+ before(): Promise<string> // content at the session baseline ('' if new / non-git)
420
+ after(): Promise<string> // current content ('' if deleted)
421
+ diff(): Promise<string> // unified patch vs the baseline
422
+ }
423
+ ```
424
+
425
+ ```typescript
426
+ import { defineChatMiddleware } from '@tanstack/ai'
427
+ import { db } from './db'
428
+
429
+ const auditMiddleware = defineChatMiddleware({
430
+ name: 'audit',
431
+ sandbox: {
432
+ onFileChange: async (ctx, e) => {
433
+ const [before, after] = await Promise.all([e.before(), e.after()])
434
+ db.log({ run: ctx.runId, path: e.path, before, after })
435
+ },
436
+ },
437
+ })
438
+ ```
439
+
440
+ **Lazy — path-only hooks pay nothing.** `before()`, `after()`, and `diff()`
441
+ are methods, not fields: each only reads the file or shells out to `git` when
442
+ called. A hook that only reads `e.path`/`e.type` (like the `onFile` logger
443
+ above) never touches the filesystem or spawns a process.
444
+
445
+ **Git session baseline.** The sandbox snapshots `git rev-parse HEAD` once at
446
+ setup as the session baseline (empty string if the workspace isn't a git repo
447
+ or has no commits). `before()` and `diff()` always diff against that same
448
+ fixed baseline for the rest of the run, so `onFileChange` reports the file's
449
+ **cumulative** change since the run started, not just the delta since the
450
+ last poll. `after()` always reads current on-disk content. None of the three
451
+ accessors throw: a deleted file resolves `after()` to `''` (it still has
452
+ `before()`); a new file resolves `before()` to `''` (it still has `after()`);
453
+ a non-git workspace resolves **both** `before()` and `after()` to `''` and
454
+ makes `diff()` fall back to a synthesized add-patch built from `after()` —
455
+ except for a `delete` event in a non-git workspace, where there's nothing to
456
+ synthesize and `diff()` resolves to `''`. In a git workspace a file git
457
+ **isn't tracking yet** (a file the agent created, and every later edit to it)
458
+ diffs empty because `git diff` ignores untracked files, so `diff()` falls
459
+ back to the same synthesized add-patch whenever the file is absent at the
460
+ baseline — a create-or-edit of an untracked file never streams an empty diff.
461
+ An empty diff for a **tracked** file (identical to the baseline) stays empty,
462
+ as it should. A **git-ignored** file is withheld: the file event still fires
463
+ (you're notified it changed) but `diff()` returns `''`, so a secret like a
464
+ `.env` never has its contents surfaced in the diff feed.
465
+
466
+ **Failures are logged, not silent.** Every git/exec/fs failure behind these
467
+ accessors (and behind the `find`-poll watcher) still falls back to `''`/an
468
+ empty snapshot, but logs first: real anomalies (a failed `git diff`, an
469
+ unreadable file, a lost `find` poll) under the `errors` category (on by
470
+ default); expected-empty conditions (a new file's `before()`, a non-git
471
+ baseline) under the `sandbox` debug category.
472
+
473
+ **Hook errors are swallowed per hook.** A throwing `sandbox` hook is caught
474
+ and logged under the `errors` category (on by default) — it cannot break the
475
+ run or stop other hooks (or the `sandbox.file` chunk) from continuing.
476
+
477
+ Source: docs/sandbox/observability.md
478
+
374
479
  ## Common Mistakes
375
480
 
376
481
  ### a. MEDIUM: Trying to modify StreamChunks in middleware
@@ -451,3 +556,4 @@ Source: docs/advanced/middleware.md
451
556
 
452
557
  - See also: **ai-core/chat-experience/SKILL.md** -- Middleware hooks into the chat lifecycle
453
558
  - See also: **ai-core/structured-outputs/SKILL.md** -- Middleware now wraps the final structured-output call; use `onStructuredOutputConfig` for JSON-Schema transforms
559
+ - See also: **ai-core/ag-ui-protocol/SKILL.md** -- Reading the `sandbox.file` / `sandbox.file.diff` `CUSTOM` chunks the sandbox runtime emits alongside these `sandbox` hooks, via `ChatStream`'s typed `KnownCustomEvent` narrowing
@@ -289,7 +289,10 @@ function ChatPage() {
289
289
  return (
290
290
  <div key={part.id}>
291
291
  <p>Approve "{part.name}"?</p>
292
- <pre>{part.arguments}</pre>
292
+ {/* `part.input` is the parsed, typed object (populated once
293
+ the arguments are complete, as they are at approval
294
+ time); `part.arguments` remains the raw JSON string. */}
295
+ <pre>{JSON.stringify(part.input, null, 2)}</pre>
293
296
  <button
294
297
  onClick={() =>
295
298
  addToolApprovalResponse({
@@ -322,6 +325,15 @@ function ChatPage() {
322
325
  }
323
326
  ```
324
327
 
328
+ > **Type-safe approval:** With typed `tools`, `part.approval` exists **only**
329
+ > on parts for tools defined with `needsApproval: true`. Tools without approval
330
+ > have no `approval` field (reading it is a compile error). For a
331
+ > tool-agnostic handler over a typed union, narrow with `'approval' in part`
332
+ > (`if (part.type === 'tool-call' && 'approval' in part && part.approval)`),
333
+ > or type a shared component against the base `ToolCallPart`. An untyped
334
+ > `useChat()` keeps `approval` on every tool-call part, which is why the
335
+ > snippet above (no `tools` generic) reads it directly.
336
+
325
337
  ### Pattern 4: Lazy Tool Discovery
326
338
 
327
339
  Set `lazy: true` on rarely-needed tools. The LLM sees their names via a synthetic