@tanstack/ai 0.39.0 → 0.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/dist/esm/activities/chat/index.d.ts +2 -2
  2. package/dist/esm/activities/chat/index.js +24 -3
  3. package/dist/esm/activities/chat/index.js.map +1 -1
  4. package/dist/esm/activities/chat/middleware/compose.d.ts +2 -2
  5. package/dist/esm/activities/chat/middleware/compose.js.map +1 -1
  6. package/dist/esm/activities/chat/middleware/index.d.ts +1 -1
  7. package/dist/esm/activities/chat/middleware/sandbox-runtime.d.ts +7 -2
  8. package/dist/esm/activities/chat/middleware/sandbox-runtime.js.map +1 -1
  9. package/dist/esm/activities/chat/middleware/types.d.ts +16 -4
  10. package/dist/esm/activities/chat/stream/processor.d.ts +18 -0
  11. package/dist/esm/activities/chat/stream/processor.js +86 -4
  12. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  13. package/dist/esm/activities/generateTranscription/index.d.ts +2 -2
  14. package/dist/esm/activities/generateTranscription/index.js.map +1 -1
  15. package/dist/esm/index.d.ts +2 -1
  16. package/dist/esm/index.js +3 -0
  17. package/dist/esm/index.js.map +1 -1
  18. package/dist/esm/types.d.ts +137 -2
  19. package/dist/esm/utilities/provider-executed.d.ts +18 -0
  20. package/dist/esm/utilities/provider-executed.js +15 -0
  21. package/dist/esm/utilities/provider-executed.js.map +1 -0
  22. package/package.json +1 -1
  23. package/skills/ai-core/adapter-configuration/SKILL.md +10 -0
  24. package/skills/ai-core/adapter-configuration/references/anthropic-adapter.md +43 -12
  25. package/skills/ai-core/ag-ui-protocol/SKILL.md +59 -0
  26. package/skills/ai-core/media-generation/SKILL.md +9 -4
  27. package/skills/ai-core/middleware/SKILL.md +91 -0
  28. package/src/activities/chat/index.ts +34 -6
  29. package/src/activities/chat/middleware/compose.ts +2 -2
  30. package/src/activities/chat/middleware/index.ts +1 -0
  31. package/src/activities/chat/middleware/sandbox-runtime.ts +4 -2
  32. package/src/activities/chat/middleware/types.ts +17 -4
  33. package/src/activities/chat/stream/processor.ts +141 -3
  34. package/src/activities/generateTranscription/index.ts +6 -2
  35. package/src/index.ts +6 -0
  36. package/src/types.ts +130 -2
  37. package/src/utilities/provider-executed.ts +32 -0
package/dist/esm/index.js CHANGED
@@ -24,6 +24,7 @@ import { convertMessagesToModelMessages, generateMessageId, modelMessageToUIMess
24
24
  import { chatParamsFromRequest, chatParamsFromRequestBody, mergeAgentTools } from "./utilities/chat-params.js";
25
25
  import { uiMessagesToWire } from "./utilities/ag-ui-wire.js";
26
26
  import { isContentPart, isContentPartArray, normalizeToolResult } from "./utilities/tool-result.js";
27
+ import { getProviderExecutedMetadata, isProviderExecutedToolCall } from "./utilities/provider-executed.js";
27
28
  import { createModel, extendAdapter } from "./extend-adapter.js";
28
29
  import { ConsoleLogger } from "./logger/console-logger.js";
29
30
  import { BatchStrategy, CompositeStrategy, ImmediateStrategy, PunctuationStrategy, WordBoundaryStrategy } from "./activities/chat/stream/strategies.js";
@@ -79,9 +80,11 @@ export {
79
80
  generateSpeech,
80
81
  generateTranscription,
81
82
  generateVideo,
83
+ getProviderExecutedMetadata,
82
84
  getVideoJobStatus,
83
85
  isContentPart,
84
86
  isContentPartArray,
87
+ isProviderExecutedToolCall,
85
88
  isStandardSchema,
86
89
  maxIterations,
87
90
  mergeAgentTools,
@@ -1 +1 @@
1
- {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
1
+ {"version":3,"file":"index.js","sources":[],"sourcesContent":[],"names":[],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;"}
@@ -96,6 +96,25 @@ export interface ToolCall<TMetadata = unknown> {
96
96
  * `@tanstack/ai-gemini` sets this to `{ thoughtSignature?: string }`. */
97
97
  metadata?: TMetadata;
98
98
  }
99
+ /**
100
+ * Convention for tool-call `metadata` that marks a call as **provider-executed**
101
+ * — run by the provider's own infrastructure (e.g. Anthropic `web_search` /
102
+ * `web_fetch` server tools) rather than by the agent loop. Adapters set
103
+ * `providerExecuted: true` so that:
104
+ *
105
+ * 1. The agent loop never tries to execute the call client-side (see
106
+ * {@link isProviderExecutedToolCall} usage in the chat engine), and
107
+ * 2. The adapter can stash the raw provider result alongside it so the call —
108
+ * and its evidence — round-trips into the next turn's request.
109
+ *
110
+ * Provider-specific payloads live under a namespaced key (e.g. `anthropic`),
111
+ * keeping this convention opaque to the framework core. The index signature
112
+ * preserves those per-adapter fields.
113
+ */
114
+ export interface ProviderExecutedToolMetadata {
115
+ providerExecuted?: boolean;
116
+ [key: string]: unknown;
117
+ }
99
118
  /**
100
119
  * Supported input modality types for multimodal content.
101
120
  * - 'text': Plain text content
@@ -252,7 +271,9 @@ export interface ToolCallPart<TMetadata = unknown> {
252
271
  /** Tool execution output (for client tools or after approval) */
253
272
  output?: any;
254
273
  /** Provider-specific metadata that round-trips with the tool call.
255
- * Typed per-adapter via `TToolCallMetadata`. */
274
+ * Typed per-adapter via `TToolCallMetadata`. May follow the
275
+ * {@link ProviderExecutedToolMetadata} convention to mark provider-executed
276
+ * server tools (e.g. Anthropic `web_search`). */
256
277
  metadata?: TMetadata;
257
278
  }
258
279
  export interface ToolResultPart {
@@ -1133,6 +1154,119 @@ export interface UIResourceEvent extends CustomEvent {
1133
1154
  meta?: Record<string, unknown>;
1134
1155
  };
1135
1156
  }
1157
+ export interface SandboxFileCustomEvent extends CustomEvent {
1158
+ name: 'sandbox.file';
1159
+ value: {
1160
+ type: 'create' | 'change' | 'delete';
1161
+ path: string;
1162
+ timestamp: number;
1163
+ };
1164
+ }
1165
+ export interface SandboxFileDiffEvent extends CustomEvent {
1166
+ name: 'sandbox.file.diff';
1167
+ value: {
1168
+ path: string;
1169
+ diff: string;
1170
+ };
1171
+ }
1172
+ export interface FileChangedEvent extends CustomEvent {
1173
+ name: 'file.changed';
1174
+ value: {
1175
+ path: string;
1176
+ diff: string;
1177
+ };
1178
+ }
1179
+ export interface SessionIdEvent extends CustomEvent {
1180
+ name: `${string}.session-id`;
1181
+ value: {
1182
+ sessionId: string;
1183
+ };
1184
+ }
1185
+ export interface CodeModeExecutionStartedEvent extends CustomEvent {
1186
+ name: 'code_mode:execution_started';
1187
+ value: {
1188
+ timestamp: number;
1189
+ codeLength: number;
1190
+ };
1191
+ }
1192
+ export interface CodeModeConsoleEvent extends CustomEvent {
1193
+ name: 'code_mode:console';
1194
+ value: {
1195
+ level: 'log' | 'warn' | 'error' | 'info';
1196
+ message: string;
1197
+ timestamp: number;
1198
+ };
1199
+ }
1200
+ export interface CodeModeExternalCallEvent extends CustomEvent {
1201
+ name: 'code_mode:external_call';
1202
+ value: {
1203
+ function: string;
1204
+ args: unknown;
1205
+ timestamp: number;
1206
+ };
1207
+ }
1208
+ export interface CodeModeExternalResultEvent extends CustomEvent {
1209
+ name: 'code_mode:external_result';
1210
+ value: {
1211
+ function: string;
1212
+ result: unknown;
1213
+ duration: number;
1214
+ };
1215
+ }
1216
+ export interface CodeModeExternalErrorEvent extends CustomEvent {
1217
+ name: 'code_mode:external_error';
1218
+ value: {
1219
+ function: string;
1220
+ error: string;
1221
+ duration: number;
1222
+ };
1223
+ }
1224
+ export interface CodeModeSkillCallEvent extends CustomEvent {
1225
+ name: 'code_mode:skill_call';
1226
+ value: {
1227
+ skill: string;
1228
+ input: unknown;
1229
+ timestamp: number;
1230
+ };
1231
+ }
1232
+ export interface CodeModeSkillResultEvent extends CustomEvent {
1233
+ name: 'code_mode:skill_result';
1234
+ value: {
1235
+ skill: string;
1236
+ result: unknown;
1237
+ duration: number;
1238
+ timestamp: number;
1239
+ };
1240
+ }
1241
+ export interface CodeModeSkillErrorEvent extends CustomEvent {
1242
+ name: 'code_mode:skill_error';
1243
+ value: {
1244
+ skill: string;
1245
+ error: string;
1246
+ duration: number;
1247
+ timestamp: number;
1248
+ };
1249
+ }
1250
+ export interface SkillRegisteredEvent extends CustomEvent {
1251
+ name: 'skill:registered';
1252
+ value: {
1253
+ id: string;
1254
+ name: string;
1255
+ description: string;
1256
+ timestamp: number;
1257
+ };
1258
+ }
1259
+ /**
1260
+ * Every CUSTOM event TanStack AI itself emits, as a discriminated union on
1261
+ * `name`. User-emitted custom events (via `emitCustomEvent` with a custom name)
1262
+ * are intentionally absent — they still flow at runtime.
1263
+ */
1264
+ export type KnownCustomEvent = SandboxFileCustomEvent | SandboxFileDiffEvent | FileChangedEvent | SessionIdEvent | CodeModeExecutionStartedEvent | CodeModeConsoleEvent | CodeModeExternalCallEvent | CodeModeExternalResultEvent | CodeModeExternalErrorEvent | CodeModeSkillCallEvent | CodeModeSkillResultEvent | CodeModeSkillErrorEvent | SkillRegisteredEvent | StructuredOutputStartEvent | StructuredOutputCompleteEvent | ApprovalRequestedEvent | ToolInputAvailableEvent | UIResourceEvent;
1265
+ /** The default chat streaming result: standard chunks plus every typed
1266
+ * framework CUSTOM event, with the `value: any` catch-all excluded so
1267
+ * literal-`name` narrowing types `value`. User-emitted custom names are typed
1268
+ * out (still flow at runtime — branch outside the name narrows or cast). */
1269
+ export type ChatStream = AsyncIterable<Exclude<StreamChunk, CustomEvent> | KnownCustomEvent>;
1136
1270
  /**
1137
1271
  * Public type for streams returned by `chat({ outputSchema, stream: true })`.
1138
1272
  *
@@ -1563,6 +1697,7 @@ export interface TTSResult {
1563
1697
  * Options for audio transcription.
1564
1698
  * These are the common options supported across providers.
1565
1699
  */
1700
+ export type TranscriptionResponseFormat = 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
1566
1701
  export interface TranscriptionOptions<TProviderOptions extends object = object> {
1567
1702
  /** The model to use for transcription */
1568
1703
  model: string;
@@ -1573,7 +1708,7 @@ export interface TranscriptionOptions<TProviderOptions extends object = object>
1573
1708
  /** An optional prompt to guide the transcription */
1574
1709
  prompt?: string;
1575
1710
  /** The format of the transcription output */
1576
- responseFormat?: 'json' | 'text' | 'srt' | 'verbose_json' | 'vtt';
1711
+ responseFormat?: TranscriptionResponseFormat;
1577
1712
  /** Model-specific options for transcription */
1578
1713
  modelOptions?: TProviderOptions;
1579
1714
  /**
@@ -0,0 +1,18 @@
1
+ import { ProviderExecutedToolMetadata } from '../types.js';
2
+ /**
3
+ * Narrow a tool call's opaque `metadata` to the provider-executed convention.
4
+ * Returns the typed metadata when the call is provider-executed, else `null`.
5
+ *
6
+ * @see ProviderExecutedToolMetadata
7
+ */
8
+ export declare function getProviderExecutedMetadata(toolCall: {
9
+ metadata?: unknown;
10
+ } | null | undefined): ProviderExecutedToolMetadata | null;
11
+ /**
12
+ * True when a tool call was executed by the provider (e.g. Anthropic
13
+ * `web_search` / `web_fetch` server tools) rather than the agent loop. Such
14
+ * calls must not be routed to client-side execution and are already "complete".
15
+ */
16
+ export declare function isProviderExecutedToolCall(toolCall: {
17
+ metadata?: unknown;
18
+ } | null | undefined): boolean;
@@ -0,0 +1,15 @@
1
+ function getProviderExecutedMetadata(toolCall) {
2
+ const metadata = toolCall?.metadata;
3
+ if (typeof metadata === "object" && metadata !== null && metadata.providerExecuted === true) {
4
+ return metadata;
5
+ }
6
+ return null;
7
+ }
8
+ function isProviderExecutedToolCall(toolCall) {
9
+ return getProviderExecutedMetadata(toolCall) !== null;
10
+ }
11
+ export {
12
+ getProviderExecutedMetadata,
13
+ isProviderExecutedToolCall
14
+ };
15
+ //# sourceMappingURL=provider-executed.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"provider-executed.js","sources":["../../../src/utilities/provider-executed.ts"],"sourcesContent":["import type { ProviderExecutedToolMetadata } from '../types'\n\n/**\n * Narrow a tool call's opaque `metadata` to the provider-executed convention.\n * Returns the typed metadata when the call is provider-executed, else `null`.\n *\n * @see ProviderExecutedToolMetadata\n */\nexport function getProviderExecutedMetadata(\n toolCall: { metadata?: unknown } | null | undefined,\n): ProviderExecutedToolMetadata | null {\n const metadata = toolCall?.metadata\n if (\n typeof metadata === 'object' &&\n metadata !== null &&\n (metadata as ProviderExecutedToolMetadata).providerExecuted === true\n ) {\n return metadata as ProviderExecutedToolMetadata\n }\n return null\n}\n\n/**\n * True when a tool call was executed by the provider (e.g. Anthropic\n * `web_search` / `web_fetch` server tools) rather than the agent loop. Such\n * calls must not be routed to client-side execution and are already \"complete\".\n */\nexport function isProviderExecutedToolCall(\n toolCall: { metadata?: unknown } | null | undefined,\n): boolean {\n return getProviderExecutedMetadata(toolCall) !== null\n}\n"],"names":[],"mappings":"AAQO,SAAS,4BACd,UACqC;AACrC,QAAM,WAAW,UAAU;AAC3B,MACE,OAAO,aAAa,YACpB,aAAa,QACZ,SAA0C,qBAAqB,MAChE;AACA,WAAO;AAAA,EACT;AACA,SAAO;AACT;AAOO,SAAS,2BACd,UACS;AACT,SAAO,4BAA4B,QAAQ,MAAM;AACnD;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.39.0",
3
+ "version": "0.40.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -297,6 +297,16 @@ Per-provider sampling keys (all live inside `modelOptions`):
297
297
  some sampling options use provider-native names. Ollama nests all sampling under
298
298
  `modelOptions.options`.
299
299
 
300
+ > **Anthropic `max_tokens` default:** Anthropic's API _requires_ `max_tokens`,
301
+ > so the adapter always sends one. When you omit `modelOptions.max_tokens`, it
302
+ > defaults to the selected model's full output ceiling (its `max_output_tokens`
303
+ > from model metadata — e.g. 64K for Sonnet, 128K for Opus), not a low constant.
304
+ > `max_tokens` is a ceiling, not a reservation (billing is per token generated),
305
+ > so leaving it unset is the right default for codegen / agentic / long-form
306
+ > output and avoids silent `stop_reason: "max_tokens"` truncation. Set it only to
307
+ > cap output below the model ceiling. Other providers treat token limits as
308
+ > optional and don't apply this flooring.
309
+
300
310
  ### 6. Capability Flag: `supportsCombinedToolsAndSchema`
301
311
 
302
312
  Adapters can declare an optional capability method:
@@ -21,17 +21,23 @@ import { anthropicText } from '@tanstack/ai-anthropic'
21
21
 
22
22
  ## Key Chat Models
23
23
 
24
- | Model | Context Window | Max Output | Notes |
25
- | ------------------- | -------------- | ---------- | ------------------------------- |
26
- | `claude-opus-4-6` | 200K | 128K | Most capable, adaptive thinking |
27
- | `claude-sonnet-4-6` | 1M | 64K | Best balance, adaptive thinking |
28
- | `claude-sonnet-4-5` | 200K | 64K | Previous gen balanced |
29
- | `claude-opus-4-5` | 200K | 32K | Previous gen most capable |
30
- | `claude-haiku-4-5` | 200K | 64K | Fast and affordable |
31
- | `claude-sonnet-4` | 200K | 64K | Older balanced model |
32
- | `claude-opus-4` | 200K | 32K | Older most capable |
33
-
34
- Note: Model IDs use the format `claude-opus-4-6`, `claude-sonnet-4-6`, etc.
24
+ | Model | Context Window | Max Output | Notes |
25
+ | ------------------- | -------------- | ---------- | ------------------------------------------- |
26
+ | `claude-fable-5` | 1M | 128K | Most capable; thinking always on (adaptive) |
27
+ | `claude-sonnet-5` | 1M | 128K | Best balance; adaptive thinking by default |
28
+ | `claude-opus-4-8` | 1M | 128K | Opus tier; adaptive thinking, no sampling |
29
+ | `claude-opus-4-7` | 1M | 128K | Older Opus; adaptive thinking, no sampling |
30
+ | `claude-opus-4-6` | 200K | 128K | Older Opus, adaptive + budget thinking |
31
+ | `claude-sonnet-4-6` | 1M | 64K | Previous gen balanced, adaptive + budget |
32
+ | `claude-sonnet-4-5` | 200K | 64K | Previous gen balanced |
33
+ | `claude-opus-4-5` | 200K | 32K | Previous gen most capable |
34
+ | `claude-opus-4-1` | 200K | 64K | Deprecated (retires 2026-08-05) |
35
+ | `claude-haiku-4-5` | 200K | 64K | Fast and affordable |
36
+
37
+ Note: Model IDs use the format `claude-sonnet-5`, `claude-opus-4-8`, etc.
38
+ Retired models (Claude 3.x, Sonnet 3.7, Opus 4 / Sonnet 4) and the `-fast`
39
+ variant ids were removed — every registered id resolves against the
40
+ first-party Anthropic API.
35
41
 
36
42
  ## Provider-Specific modelOptions
37
43
 
@@ -90,11 +96,36 @@ chat({
90
96
  ANTHROPIC_API_KEY
91
97
  ```
92
98
 
99
+ ## Adaptive-era modelOptions (Sonnet 5, Fable 5, Opus 4.7/4.8)
100
+
101
+ The per-model types restrict `modelOptions` on the newest models:
102
+
103
+ ```typescript
104
+ chat({
105
+ adapter: anthropicText('claude-sonnet-5'), // or 'claude-fable-5', 'claude-opus-4-8'
106
+ messages,
107
+ modelOptions: {
108
+ // Adaptive thinking only — budget_tokens is rejected (400).
109
+ // On claude-fable-5, { type: 'disabled' } is also rejected;
110
+ // elsewhere it opts out of thinking.
111
+ thinking: { type: 'adaptive', display: 'summarized' },
112
+ // Effort lives under output_config; 'xhigh' is available on
113
+ // Opus 4.7+, Sonnet 5, and Fable 5.
114
+ output_config: { effort: 'xhigh' },
115
+ max_tokens: 64_000,
116
+ // NO temperature / top_p / top_k — the API rejects them on these models
117
+ },
118
+ })
119
+ ```
120
+
93
121
  ## Gotchas
94
122
 
95
123
  - `thinking.budget_tokens` must be >= 1024 AND less than `modelOptions.max_tokens`.
96
124
  Failing either check throws a validation error.
97
125
  - Cannot set both `top_p` and `temperature` at the same time (throws error).
98
- - `claude-3-5-haiku` and `claude-3-haiku` do NOT support extended thinking.
126
+ - `claude-sonnet-5`, `claude-fable-5`, `claude-opus-4-8`, and
127
+ `claude-opus-4-7` do NOT accept `temperature`, `top_p`, `top_k`, or
128
+ `thinking: { type: 'enabled', budget_tokens }` — adaptive thinking +
129
+ `output_config.effort` replace them (typed per model).
99
130
  - System prompts support prompt caching via `cache_control` on `TextBlockParam[]`.
100
131
  - All Claude models accept `text`, `image`, and `document` (PDF) input.
@@ -13,6 +13,7 @@ sources:
13
13
  - 'TanStack/ai:docs/protocol/chunk-definitions.md'
14
14
  - 'TanStack/ai:docs/protocol/sse-protocol.md'
15
15
  - 'TanStack/ai:docs/protocol/http-stream-protocol.md'
16
+ - 'TanStack/ai:docs/protocol/custom-events.md'
16
17
  ---
17
18
 
18
19
  # AG-UI Protocol
@@ -218,6 +219,62 @@ RUN_STARTED -> TEXT_MESSAGE_START -> TEXT_MESSAGE_CONTENT* -> TEXT_MESSAGE_END
218
219
  union of all event interfaces). `StreamChunkType` is an alias for `AGUIEventType`
219
220
  (the string union of all event type literals).
220
221
 
222
+ ### 4. Typed CUSTOM Events — `ChatStream` and `KnownCustomEvent`
223
+
224
+ The `CUSTOM` row above describes the raw `StreamChunk` union, where the single
225
+ generic `CustomEvent` member types `value` as `any` -- once merged into a
226
+ union, that `any` poisons every other member too, so narrowing on `name`
227
+ still leaves `value: any`. `chat()` doesn't return raw `StreamChunk`; by
228
+ default (no `outputSchema`, `stream` not explicitly `false`) it returns
229
+ `ChatStream`, which swaps that generic member for `KnownCustomEvent` -- a
230
+ discriminated union of every `CUSTOM` event TanStack AI itself emits, each
231
+ with a literal `name` and a concrete `value`. Narrow with a plain `if` --
232
+ no helper, no cast:
233
+
234
+ ```typescript
235
+ import { chat } from '@tanstack/ai'
236
+ import { openaiText } from '@tanstack/ai-openai'
237
+
238
+ const stream = chat({
239
+ adapter: openaiText('gpt-5.2'),
240
+ messages,
241
+ })
242
+
243
+ for await (const chunk of stream) {
244
+ if (chunk.type === 'CUSTOM' && chunk.name === 'sandbox.file.diff') {
245
+ console.log(chunk.value.path, chunk.value.diff) // typed, no helper, no cast
246
+ } else if (
247
+ chunk.type === 'CUSTOM' &&
248
+ chunk.name === 'structured-output.complete'
249
+ ) {
250
+ console.log(chunk.value.object) // typed, no helper, no cast
251
+ }
252
+ }
253
+ ```
254
+
255
+ **Caveat -- `.endsWith()` (or any non-literal check) does not narrow.**
256
+ `SessionIdEvent['name']` is the template-literal type
257
+ `` `${string}.session-id` ``. TypeScript's control-flow narrowing only
258
+ understands exact comparisons (`===`) and `in`/type-predicate checks against
259
+ a discriminant -- a runtime `chunk.name.endsWith('.session-id')` check
260
+ doesn't inform the type system, so `chunk.value` stays the union of every
261
+ `KnownCustomEvent`'s `value`, not `{ sessionId: string }`. Compare against
262
+ the exact literal you expect, or write a user-defined type predicate
263
+ (`(c): c is SessionIdEvent => c.name.endsWith('.session-id')`) and call that
264
+ in the `if` instead.
265
+
266
+ **User-emitted `emitCustomEvent` names are typed out of `ChatStream`.** Tools
267
+ that call `context.emitCustomEvent('my-app:progress', ...)` still stream a
268
+ `CUSTOM` chunk at runtime, but `'my-app:progress'` isn't one of
269
+ `KnownCustomEvent`'s literal names, so it's intentionally absent from
270
+ `ChatStream`'s type -- including a generic fallback member would reintroduce
271
+ the `value: any` poison for every other event on the stream. To read your own
272
+ event with a type, annotate the stream as the wider `StreamChunk` instead of
273
+ `ChatStream` for that branch; its generic `CUSTOM` member already types
274
+ `value` as `any`, so no cast is needed there either.
275
+
276
+ Source: docs/protocol/custom-events.md
277
+
221
278
  ## Common Mistakes
222
279
 
223
280
  ### MEDIUM: Proxy buffering breaks SSE streaming
@@ -273,3 +330,5 @@ without transformation. See `docs/migration/ag-ui-compliance.md` for details.
273
330
  ## Cross-References
274
331
 
275
332
  - See also: `ai-core/custom-backend-integration/SKILL.md` -- Custom backends must implement SSE or HTTP stream format to work with TanStack AI client connection adapters.
333
+ - See also: `ai-core/middleware/SKILL.md` -- `sandbox.file.diff`'s `{ path, diff }` value (one of `KnownCustomEvent`'s members) is populated from the same lazy `before()`/`after()`/`diff()` accessors documented there for `onFile*` middleware hooks.
334
+ - Full CUSTOM event taxonomy: `docs/protocol/custom-events.md`.
@@ -151,7 +151,7 @@ function ImageGenerator() {
151
151
 
152
152
  Supported adapters: `openaiImage` (dall-e-2, dall-e-3, gpt-image-1,
153
153
  gpt-image-1-mini, gpt-image-2) and `geminiImage` (gemini-3.1-flash-image-preview,
154
- imagen-4.0-generate-001, etc.).
154
+ gemini-3.1-flash-lite-image, imagen-4.0-generate-001, etc.).
155
155
 
156
156
  ```typescript
157
157
  import { generateImage } from '@tanstack/ai'
@@ -357,7 +357,7 @@ const { generate, result, isLoading } = useGenerateSpeech({
357
357
  ### 4. Audio Transcription
358
358
 
359
359
  Adapter: `openaiTranscription` (whisper-1, gpt-4o-transcribe,
360
- gpt-4o-mini-transcribe).
360
+ gpt-4o-mini-transcribe, gpt-4o-transcribe-diarize).
361
361
 
362
362
  > **Capturing audio in the browser:** Use `useAudioRecorder` from `@tanstack/ai-react` to record directly in the browser, then pass the recording as the `audio` input to `generate()`, or use `recording.part` as a prompt part in chat/generation calls. No transcoding or extra dependencies required — the recorder returns the native browser format (`audio/webm` or `audio/mp4`). For transcription, wrap it as a `data:` URL so the provider gets the real content type; passing raw `recording.base64` makes the adapter assume `audio/mpeg` and mislabel the webm/mp4 bytes.
363
363
  >
@@ -382,16 +382,21 @@ const result = await generateTranscription({
382
382
  language: 'en',
383
383
  responseFormat: 'verbose_json',
384
384
  modelOptions: {
385
- include: ['segment', 'word'],
385
+ timestamp_granularities: ['word', 'segment'],
386
386
  },
387
387
  })
388
388
 
389
389
  // result.text -- full transcribed text
390
390
  // result.language -- detected/specified language
391
391
  // result.duration -- audio duration in seconds
392
- // result.segments -- timestamped segments with optional word-level timestamps
392
+ // result.segments -- timestamped segments (word-level timestamps are in result.words)
393
393
  ```
394
394
 
395
+ For speaker diarization, use `openaiTranscription('gpt-4o-transcribe-diarize')`.
396
+ When no response format is given it defaults the request to `response_format: 'diarized_json'`
397
+ and `chunking_strategy: 'auto'` (a top-level `responseFormat` of `'json'`/`'text'` opts out of
398
+ speaker segments); do not pass `prompt`, `include`, or `timestamp_granularities` with this model.
399
+
395
400
  Client hook:
396
401
 
397
402
  ```tsx
@@ -11,6 +11,7 @@ library: tanstack-ai
11
11
  library_version: '0.10.0'
12
12
  sources:
13
13
  - 'TanStack/ai:docs/advanced/middleware.md'
14
+ - 'TanStack/ai:docs/sandbox/observability.md'
14
15
  ---
15
16
 
16
17
  # Middleware
@@ -371,6 +372,95 @@ Options: `maxSize` (default 100), `ttl` (default Infinity), `toolNames` (default
371
372
  `keyFn` (custom cache key), `storage` (custom backend like Redis). See
372
373
  `docs/advanced/middleware.md` for custom storage examples.
373
374
 
375
+ ## Sandbox File-Event Hooks (`sandbox` group)
376
+
377
+ Declare a `sandbox: ChatSandboxHooks` group on `defineChatMiddleware` to react
378
+ to every file created/changed/deleted inside a sandbox provided by
379
+ `withSandbox` (from `@tanstack/ai-sandbox`). These fire **per-run**,
380
+ server-side, and each handler receives the run's `ChatMiddlewareContext` as
381
+ the first argument:
382
+
383
+ ```typescript
384
+ import { defineChatMiddleware } from '@tanstack/ai'
385
+ import { db } from './db'
386
+
387
+ const auditMiddleware = defineChatMiddleware({
388
+ name: 'audit',
389
+ sandbox: {
390
+ onFile: (ctx, e) => console.log(ctx.runId, e.type, e.path),
391
+ onFileCreate: (ctx, e) => db.log({ run: ctx.runId, event: e }),
392
+ },
393
+ })
394
+ ```
395
+
396
+ | Hook | Fires for |
397
+ | -------------- | -------------------------- |
398
+ | `onFile` | Every create/change/delete |
399
+ | `onFileCreate` | File creates only |
400
+ | `onFileChange` | File changes only |
401
+ | `onFileDelete` | File deletes only |
402
+
403
+ These are independent of the stream: the engine also emits a `sandbox.file`
404
+ `CUSTOM` chunk per change regardless of whether any `sandbox` hooks are
405
+ registered, so a client can react to the same edits without middleware. See
406
+ `ai-core/ag-ui-protocol/SKILL.md` for reading that chunk (and the opt-in
407
+ `sandbox.file.diff` chunk) off `ChatStream`.
408
+
409
+ ### `before()` / `after()` / `diff()` — lazy, git-backed content accessors
410
+
411
+ Each hook receives a `SandboxFileHookEvent`: the serializable
412
+ `{ type, path, timestamp }` plus three lazy accessors for the file's content:
413
+
414
+ ```ts
415
+ interface SandboxFileHookEvent {
416
+ type: 'create' | 'change' | 'delete'
417
+ path: string
418
+ timestamp: number
419
+ before(): Promise<string> // content at the session baseline ('' if new / non-git)
420
+ after(): Promise<string> // current content ('' if deleted)
421
+ diff(): Promise<string> // unified patch vs the baseline
422
+ }
423
+ ```
424
+
425
+ ```typescript
426
+ import { defineChatMiddleware } from '@tanstack/ai'
427
+ import { db } from './db'
428
+
429
+ const auditMiddleware = defineChatMiddleware({
430
+ name: 'audit',
431
+ sandbox: {
432
+ onFileChange: async (ctx, e) => {
433
+ const [before, after] = await Promise.all([e.before(), e.after()])
434
+ db.log({ run: ctx.runId, path: e.path, before, after })
435
+ },
436
+ },
437
+ })
438
+ ```
439
+
440
+ **Lazy — path-only hooks pay nothing.** `before()`, `after()`, and `diff()`
441
+ are methods, not fields: each only reads the file or shells out to `git` when
442
+ called. A hook that only reads `e.path`/`e.type` (like the `onFile` logger
443
+ above) never touches the filesystem or spawns a process.
444
+
445
+ **Git session baseline.** The sandbox snapshots `git rev-parse HEAD` once at
446
+ setup as the session baseline (empty string if the workspace isn't a git repo
447
+ or has no commits). `before()` and `diff()` always diff against that same
448
+ fixed baseline for the rest of the run, so `onFileChange` reports the file's
449
+ **cumulative** change since the run started, not just the delta since the
450
+ last poll. `after()` always reads current on-disk content. None of the three
451
+ accessors throw: a deleted file resolves `after()` to `''` (it still has
452
+ `before()`); a new file resolves `before()` to `''` (it still has `after()`);
453
+ a non-git workspace resolves **both** `before()` and `after()` to `''` and
454
+ makes `diff()` fall back to a synthesized add-patch built from `after()` —
455
+ except for a `delete` event in a non-git workspace, where there's nothing to
456
+ synthesize and `diff()` resolves to `''`.
457
+
458
+ **Hook errors are swallowed per hook.** A throwing `sandbox` hook is caught
459
+ and logged under the `sandbox` debug category — it cannot break the run or
460
+ stop other hooks (or the `sandbox.file` chunk) from continuing.
461
+
462
+ Source: docs/sandbox/observability.md
463
+
374
464
  ## Common Mistakes
375
465
 
376
466
  ### a. MEDIUM: Trying to modify StreamChunks in middleware
@@ -451,3 +541,4 @@ Source: docs/advanced/middleware.md
451
541
 
452
542
  - See also: **ai-core/chat-experience/SKILL.md** -- Middleware hooks into the chat lifecycle
453
543
  - See also: **ai-core/structured-outputs/SKILL.md** -- Middleware now wraps the final structured-output call; use `onStructuredOutputConfig` for JSON-Schema transforms
544
+ - See also: **ai-core/ag-ui-protocol/SKILL.md** -- Reading the `sandbox.file` / `sandbox.file.diff` `CUSTOM` chunks the sandbox runtime emits alongside these `sandbox` hooks, via `ChatStream`'s typed `KnownCustomEvent` narrowing
@@ -12,6 +12,7 @@ import { streamToText } from '../../stream-to-response.js'
12
12
  import { resolveDebugOption } from '../../logger/resolve'
13
13
  import { EventType } from '../../types'
14
14
  import { normalizeToolResult } from '../../utilities/tool-result'
15
+ import { isProviderExecutedToolCall } from '../../utilities/provider-executed'
15
16
  import { LazyToolManager } from './tools/lazy-tool-manager'
16
17
  import {
17
18
  MiddlewareAbortError,
@@ -44,6 +45,7 @@ import type {
44
45
  import type {
45
46
  AgentLoopStrategy,
46
47
  AnyTool,
48
+ ChatStream,
47
49
  ConstrainedModelMessage,
48
50
  CustomEvent,
49
51
  InferSchemaType,
@@ -68,7 +70,7 @@ import type {
68
70
  ChatMiddleware,
69
71
  ChatMiddlewareConfig,
70
72
  ChatMiddlewareContext,
71
- SandboxFileEvent,
73
+ SandboxFileHookEvent,
72
74
  StructuredOutputMiddlewareConfig,
73
75
  } from './middleware/types'
74
76
  import type { CheckCoverage } from './middleware/builder'
@@ -403,7 +405,7 @@ export type TextActivityResult<
403
405
  : Promise<InferSchemaType<TSchema>>
404
406
  : [TStream] extends [false]
405
407
  ? Promise<string>
406
- : AsyncIterable<StreamChunk>
408
+ : ChatStream
407
409
 
408
410
  // ===========================
409
411
  // ChatEngine Implementation
@@ -711,11 +713,30 @@ class TextEngine<
711
713
  // a `sandbox.file` custom chunk to be drained into the public stream.
712
714
  provideSandboxRuntime(this.middlewareCtx, {
713
715
  logger: this.logger,
714
- emit: (event: SandboxFileEvent) => {
715
- this.logger.sandbox(`file ${event.type} ${event.path}`, { event })
716
- void this.middlewareRunner.runSandboxFile(this.middlewareCtx, event)
716
+ emit: (event: SandboxFileHookEvent) => {
717
+ this.logger.sandbox(`file ${event.type} ${event.path}`, {
718
+ event: {
719
+ type: event.type,
720
+ path: event.path,
721
+ timestamp: event.timestamp,
722
+ },
723
+ })
724
+ void this.middlewareRunner
725
+ .runSandboxFile(this.middlewareCtx, event)
726
+ .catch((err: unknown) => {
727
+ this.logger.errors('sandbox file hook failed', { error: err })
728
+ })
717
729
  this.sandboxFileQueue.push(
718
- this.createCustomEventChunk('sandbox.file', { ...event }),
730
+ this.createCustomEventChunk('sandbox.file', {
731
+ type: event.type,
732
+ path: event.path,
733
+ timestamp: event.timestamp,
734
+ }),
735
+ )
736
+ },
737
+ emitFileDiff: (value: { path: string; diff: string }) => {
738
+ this.sandboxFileQueue.push(
739
+ this.createCustomEventChunk('sandbox.file.diff', value),
719
740
  )
720
741
  },
721
742
  })
@@ -1884,6 +1905,13 @@ class TextEngine<
1884
1905
  for (const message of this.messages) {
1885
1906
  if (message.role === 'assistant' && message.toolCalls) {
1886
1907
  for (const toolCall of message.toolCalls) {
1908
+ // Provider-executed tool calls (e.g. Anthropic `web_search`) were
1909
+ // already run by the provider; they carry no client result, so they
1910
+ // would otherwise look "pending" forever and the loop would try (and
1911
+ // fail) to execute them client-side. Skip them.
1912
+ if (isProviderExecutedToolCall(toolCall)) {
1913
+ continue
1914
+ }
1887
1915
  if (!completedToolIds.has(toolCall.id)) {
1888
1916
  pending.push(toolCall)
1889
1917
  }
@@ -11,7 +11,7 @@ import type {
11
11
  ErrorInfo,
12
12
  FinishInfo,
13
13
  IterationInfo,
14
- SandboxFileEvent,
14
+ SandboxFileHookEvent,
15
15
  StructuredOutputMiddlewareConfig,
16
16
  ToolCallHookContext,
17
17
  ToolPhaseCompleteInfo,
@@ -352,7 +352,7 @@ export class MiddlewareRunner<TContext = unknown> {
352
352
  */
353
353
  async runSandboxFile(
354
354
  ctx: ChatMiddlewareContext<TContext>,
355
- event: SandboxFileEvent,
355
+ event: SandboxFileHookEvent,
356
356
  ): Promise<void> {
357
357
  const typed = (
358
358
  {