@tanstack/ai 0.31.0 → 0.33.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (83) hide show
  1. package/dist/esm/activities/chat/index.js +24 -3
  2. package/dist/esm/activities/chat/index.js.map +1 -1
  3. package/dist/esm/activities/chat/middleware/types.d.ts +7 -0
  4. package/dist/esm/activities/chat/tools/lazy-tool-manager.d.ts +25 -1
  5. package/dist/esm/activities/chat/tools/lazy-tool-manager.js +26 -2
  6. package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
  7. package/dist/esm/activities/generateAudio/index.d.ts +7 -0
  8. package/dist/esm/activities/generateAudio/index.js +26 -1
  9. package/dist/esm/activities/generateAudio/index.js.map +1 -1
  10. package/dist/esm/activities/generateImage/adapter.d.ts +8 -4
  11. package/dist/esm/activities/generateImage/adapter.js.map +1 -1
  12. package/dist/esm/activities/generateImage/index.d.ts +26 -3
  13. package/dist/esm/activities/generateImage/index.js +38 -2
  14. package/dist/esm/activities/generateImage/index.js.map +1 -1
  15. package/dist/esm/activities/generateSpeech/index.d.ts +7 -0
  16. package/dist/esm/activities/generateSpeech/index.js +26 -1
  17. package/dist/esm/activities/generateSpeech/index.js.map +1 -1
  18. package/dist/esm/activities/generateTranscription/index.d.ts +7 -0
  19. package/dist/esm/activities/generateTranscription/index.js +26 -1
  20. package/dist/esm/activities/generateTranscription/index.js.map +1 -1
  21. package/dist/esm/activities/generateVideo/adapter.d.ts +65 -6
  22. package/dist/esm/activities/generateVideo/adapter.js +14 -0
  23. package/dist/esm/activities/generateVideo/adapter.js.map +1 -1
  24. package/dist/esm/activities/generateVideo/index.d.ts +40 -5
  25. package/dist/esm/activities/generateVideo/index.js +52 -2
  26. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  27. package/dist/esm/activities/generateVideo/snap.d.ts +14 -0
  28. package/dist/esm/activities/generateVideo/snap.js +54 -0
  29. package/dist/esm/activities/generateVideo/snap.js.map +1 -0
  30. package/dist/esm/activities/index.d.ts +3 -2
  31. package/dist/esm/activities/index.js +2 -0
  32. package/dist/esm/activities/index.js.map +1 -1
  33. package/dist/esm/activities/middleware/index.d.ts +2 -0
  34. package/dist/esm/activities/middleware/run.d.ts +20 -0
  35. package/dist/esm/activities/middleware/run.js +42 -0
  36. package/dist/esm/activities/middleware/run.js.map +1 -0
  37. package/dist/esm/activities/middleware/types.d.ts +118 -0
  38. package/dist/esm/client.d.ts +1 -1
  39. package/dist/esm/client.js.map +1 -1
  40. package/dist/esm/index.d.ts +4 -0
  41. package/dist/esm/index.js +4 -0
  42. package/dist/esm/index.js.map +1 -1
  43. package/dist/esm/middlewares/otel.d.ts +8 -2
  44. package/dist/esm/middlewares/otel.js +145 -95
  45. package/dist/esm/middlewares/otel.js.map +1 -1
  46. package/dist/esm/middlewares/usage-attributes.d.ts +24 -0
  47. package/dist/esm/middlewares/usage-attributes.js +43 -0
  48. package/dist/esm/middlewares/usage-attributes.js.map +1 -0
  49. package/dist/esm/types.d.ts +103 -14
  50. package/dist/esm/utilities/errors.d.ts +13 -0
  51. package/dist/esm/utilities/errors.js +22 -0
  52. package/dist/esm/utilities/errors.js.map +1 -0
  53. package/dist/esm/utilities/media-prompt.d.ts +35 -0
  54. package/dist/esm/utilities/media-prompt.js +43 -0
  55. package/dist/esm/utilities/media-prompt.js.map +1 -0
  56. package/dist/esm/utilities/numbers.d.ts +8 -0
  57. package/dist/esm/utilities/numbers.js +12 -0
  58. package/dist/esm/utilities/numbers.js.map +1 -0
  59. package/package.json +2 -2
  60. package/skills/ai-core/media-generation/SKILL.md +173 -3
  61. package/src/activities/chat/index.ts +32 -4
  62. package/src/activities/chat/middleware/types.ts +7 -0
  63. package/src/activities/chat/tools/lazy-tool-manager.ts +46 -4
  64. package/src/activities/generateAudio/index.ts +42 -1
  65. package/src/activities/generateImage/adapter.ts +16 -3
  66. package/src/activities/generateImage/index.ts +90 -5
  67. package/src/activities/generateSpeech/index.ts +42 -1
  68. package/src/activities/generateTranscription/index.ts +42 -1
  69. package/src/activities/generateVideo/adapter.ts +80 -4
  70. package/src/activities/generateVideo/index.ts +141 -6
  71. package/src/activities/generateVideo/snap.ts +100 -0
  72. package/src/activities/index.ts +4 -0
  73. package/src/activities/middleware/index.ts +20 -0
  74. package/src/activities/middleware/run.ts +88 -0
  75. package/src/activities/middleware/types.ts +173 -0
  76. package/src/client.ts +4 -0
  77. package/src/index.ts +23 -0
  78. package/src/middlewares/otel.ts +195 -120
  79. package/src/middlewares/usage-attributes.ts +65 -0
  80. package/src/types.ts +126 -13
  81. package/src/utilities/errors.ts +29 -0
  82. package/src/utilities/media-prompt.ts +86 -0
  83. package/src/utilities/numbers.ts +15 -0
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Return the first candidate that is a finite `number`, or `undefined`.
3
+ *
4
+ * Handy for picking a value from among several possible spellings/sources where
5
+ * only some are populated — e.g. the provider-native sampling option names read
6
+ * by the OTel middleware, or the optional numeric fields on `TokenUsage`.
7
+ */
8
+ export declare function firstNumber(...candidates: Array<unknown>): number | undefined;
@@ -0,0 +1,12 @@
1
+ function firstNumber(...candidates) {
2
+ for (const candidate of candidates) {
3
+ if (typeof candidate === "number" && Number.isFinite(candidate)) {
4
+ return candidate;
5
+ }
6
+ }
7
+ return void 0;
8
+ }
9
+ export {
10
+ firstNumber
11
+ };
12
+ //# sourceMappingURL=numbers.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"numbers.js","sources":["../../../src/utilities/numbers.ts"],"sourcesContent":["/**\n * Return the first candidate that is a finite `number`, or `undefined`.\n *\n * Handy for picking a value from among several possible spellings/sources where\n * only some are populated — e.g. the provider-native sampling option names read\n * by the OTel middleware, or the optional numeric fields on `TokenUsage`.\n */\nexport function firstNumber(...candidates: Array<unknown>): number | undefined {\n for (const candidate of candidates) {\n if (typeof candidate === 'number' && Number.isFinite(candidate)) {\n return candidate\n }\n }\n return undefined\n}\n"],"names":[],"mappings":"AAOO,SAAS,eAAe,YAAgD;AAC7E,aAAW,aAAa,YAAY;AAClC,QAAI,OAAO,cAAc,YAAY,OAAO,SAAS,SAAS,GAAG;AAC/D,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.31.0",
3
+ "version": "0.33.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -76,7 +76,7 @@
76
76
  "@ag-ui/core": "^0.0.52",
77
77
  "@standard-schema/spec": "^1.1.0",
78
78
  "partial-json": "^0.1.7",
79
- "@tanstack/ai-event-client": "0.6.2"
79
+ "@tanstack/ai-event-client": "0.6.4"
80
80
  },
81
81
  "peerDependencies": {
82
82
  "@opentelemetry/api": ">=1.9.0"
@@ -3,8 +3,9 @@ name: ai-core/media-generation
3
3
  description: >
4
4
  Image, audio, video, speech (TTS), and transcription generation using
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage,
6
- generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
- polling, generateSpeech() with openaiSpeech, generateTranscription() with
6
+ generateAudio() with geminiAudio/falAudio, generateVideo() with
7
+ openaiVideo/geminiVideo (async polling, per-model typed durations),
8
+ generateSpeech() with openaiSpeech, generateTranscription() with
8
9
  openaiTranscription. React hooks: useGenerateImage, useGenerateAudio,
9
10
  useGenerateSpeech, useTranscription, useGenerateVideo.
10
11
  TanStack Start server function integration with toServerSentEventsResponse.
@@ -189,6 +190,103 @@ Result shape: `ImageGenerationResult` with `images` array where each entry
189
190
  has `b64Json?`, `url?`, and `revisedPrompt?`. OpenAI image URLs expire
190
191
  after 1 hour -- download or display immediately.
191
192
 
193
+ #### Image-conditioned generation: multimodal `prompt` parts
194
+
195
+ Both `generateImage()` and `generateVideo()` accept the `prompt` either as
196
+ a plain string or as an ordered array of content parts (`TextPart` /
197
+ `ImagePart` / `VideoPart` / `AudioPart` — the same shapes used elsewhere in
198
+ TanStack AI). Part order is meaningful: natively multimodal providers
199
+ (Gemini, OpenRouter) receive parts in order; named-field providers (OpenAI,
200
+ fal, xAI) extract media parts and flatten the text. Prompt text is always
201
+ sent verbatim — to reference inputs from the prompt, write the provider's
202
+ own syntax (fal `@Image1`, OpenAI "image 1" prose); the SDK never injects
203
+ or rewrites markers. Each media part may carry an optional
204
+ `metadata.role` hint that adapters use to route the part to the
205
+ provider-specific field. The accepted part types are narrowed per model at
206
+ compile time via the adapter's input-modality map.
207
+
208
+ ```typescript
209
+ import { generateImage } from '@tanstack/ai'
210
+ import { openaiImage } from '@tanstack/ai-openai'
211
+
212
+ // Image-to-image (OpenAI gpt-image-2 / gpt-image-1, dall-e-2)
213
+ await generateImage({
214
+ adapter: openaiImage('gpt-image-2'),
215
+ prompt: [
216
+ { type: 'text', content: 'Turn this into a cinematic product photo' },
217
+ { type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
218
+ ],
219
+ })
220
+
221
+ // Multi-reference (up to 16 for gpt-image models; up to ~14 for Gemini native
222
+ // — a provider limit, not enforced by the SDK)
223
+ await generateImage({
224
+ adapter: openaiImage('gpt-image-2'),
225
+ prompt: [
226
+ { type: 'text', content: 'Apply the second image as style to the first' },
227
+ { type: 'image', source: { type: 'url', value: 'https://…/product.png' } },
228
+ { type: 'image', source: { type: 'url', value: 'https://…/style.png' } },
229
+ ],
230
+ })
231
+
232
+ // Inpaint via metadata.role === 'mask' (OpenAI gpt-image models, dall-e-2; fal mask_url)
233
+ await generateImage({
234
+ adapter: openaiImage('gpt-image-2'),
235
+ prompt: [
236
+ { type: 'text', content: 'Replace the masked region with a tree' },
237
+ { type: 'image', source: { type: 'url', value: photoUrl } },
238
+ {
239
+ type: 'image',
240
+ source: { type: 'url', value: maskUrl },
241
+ metadata: { role: 'mask' },
242
+ },
243
+ ],
244
+ })
245
+
246
+ // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
247
+ import { generateVideo } from '@tanstack/ai'
248
+ import { falVideo } from '@tanstack/ai-fal'
249
+
250
+ await generateVideo({
251
+ adapter: falVideo('fal-ai/kling-video/v3/pro/image-to-video'),
252
+ prompt: [
253
+ { type: 'image', source: { type: 'url', value: firstFrameUrl } },
254
+ { type: 'text', content: 'Slow cinematic push-in' },
255
+ {
256
+ type: 'image',
257
+ source: { type: 'url', value: lastFrameUrl },
258
+ metadata: { role: 'end_frame' },
259
+ },
260
+ ],
261
+ })
262
+ ```
263
+
264
+ **Role hints** (`metadata.role`):
265
+
266
+ | Role | Maps to |
267
+ | --------------- | ----------------------------------------------------------------------------------------------------- |
268
+ | `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
269
+ | `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
270
+ | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
271
+ | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
272
+ | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
273
+ | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
274
+
275
+ **Provider support matrix:**
276
+
277
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
278
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
279
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
280
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
281
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
282
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
283
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
284
+ | Anthropic | n/a (no image generation API). | n/a |
285
+
286
+ Video and audio prompt parts follow the same `metadata.role` convention
287
+ for video-to-video and lipsync flows on fal; other providers throw when
288
+ they're passed.
289
+
192
290
  ### 2. Audio Generation (Music, Sound Effects)
193
291
 
194
292
  Distinct from TTS — `generateAudio()` produces non-speech audio content.
@@ -331,6 +429,31 @@ const stream = generateVideo({
331
429
  return toServerSentEventsResponse(stream)
332
430
  ```
333
431
 
432
+ Google Veo (`@tanstack/ai-gemini`) uses the same jobs/polling flow. Its
433
+ `duration` option is typed per model (e.g. `4 | 6 | 8` for Veo 3.x,
434
+ `5 | 6 | 8` for Veo 2); use `adapter.snapDuration(seconds)` to coerce raw
435
+ seconds and `adapter.availableDurations()` to enumerate the valid set.
436
+ Image prompt parts route by `metadata.role`: first un-roled /
437
+ `'start_frame'` image → input image, `'end_frame'` → `lastFrame`,
438
+ `'reference'` / `'character'` → `referenceImages`:
439
+
440
+ ```typescript
441
+ import { geminiVideo } from '@tanstack/ai-gemini'
442
+
443
+ const adapter = geminiVideo('veo-3.1-generate-preview')
444
+ adapter.availableDurations() // { kind: 'discrete', values: [4, 6, 8] }
445
+
446
+ const { jobId } = await generateVideo({
447
+ adapter,
448
+ prompt: 'A golden retriever playing in sunflowers',
449
+ size: '16:9', // Veo sizes are aspect ratios: '16:9' | '9:16'
450
+ duration: adapter.snapDuration(7), // 6
451
+ modelOptions: { resolution: '1080p', generateAudio: true },
452
+ })
453
+ // Note: Veo result URLs require the Google API key to download
454
+ // (x-goog-api-key header or ?key= query parameter).
455
+ ```
456
+
334
457
  Client hook with job tracking:
335
458
 
336
459
  ```tsx
@@ -607,7 +730,54 @@ generateSpeech({
607
730
 
608
731
  > Source: Gemini TTS adapter validation; CodeRabbit review of PR #463.
609
732
 
610
- ### h. LOW: Writing a logging middleware to see media chunks flow through
733
+ ### h. HIGH: Passing image prompt parts to a model that doesn't support image-conditioned generation
734
+
735
+ Not every model accepts image-conditioned prompts. The `prompt` type is
736
+ narrowed per model, so passing an image part to a text-only model
737
+ (dall-e-3, Imagen, grok-2-image) is a **compile-time error**; adapters
738
+ also throw a clear runtime error as a backstop, so users learn at call
739
+ time rather than getting silently wrong output.
740
+
741
+ ```typescript
742
+ // WRONG — dall-e-3 has no edit/inputs API; image parts are a type error
743
+ generateImage({
744
+ adapter: openaiImage('dall-e-3'),
745
+ prompt: [
746
+ { type: 'text', content: 'Edit this' },
747
+ { type: 'image', source: { type: 'url', value: url } }, // ❌ type error
748
+ ],
749
+ })
750
+
751
+ // WRONG — Imagen is text-to-image only; same compile-time rejection
752
+ generateImage({
753
+ adapter: geminiImage('imagen-4.0-generate-001'),
754
+ prompt: [
755
+ { type: 'text', content: 'Edit this' },
756
+ { type: 'image', source: { type: 'url', value: url } }, // ❌ type error
757
+ ],
758
+ })
759
+
760
+ // CORRECT — use a model that supports image-conditioned generation
761
+ generateImage({
762
+ adapter: openaiImage('gpt-image-2'), // edits up to 16 images
763
+ prompt: [
764
+ { type: 'text', content: 'Edit this' },
765
+ { type: 'image', source: { type: 'url', value: url } },
766
+ ],
767
+ })
768
+
769
+ generateImage({
770
+ adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
771
+ prompt: [
772
+ { type: 'text', content: 'Edit this' },
773
+ { type: 'image', source: { type: 'url', value: url } },
774
+ ],
775
+ })
776
+ ```
777
+
778
+ > Source: docs/media/image-generation.md, docs/media/video-generation.md.
779
+
780
+ ### i. LOW: Writing a logging middleware to see media chunks flow through
611
781
 
612
782
  Every media activity — `generateAudio`, `generateSpeech`,
613
783
  `generateTranscription`, `generateImage`, `generateVideo` — accepts the
@@ -33,7 +33,11 @@ import type {
33
33
  ClientToolRequest,
34
34
  ToolResult,
35
35
  } from './tools/tool-calls'
36
- import type { AnyTextAdapter, StructuredOutputOptions } from './adapter'
36
+ import type {
37
+ AnyTextAdapter,
38
+ StructuredOutputOptions,
39
+ StructuredOutputResult,
40
+ } from './adapter'
37
41
  import type {
38
42
  AgentLoopStrategy,
39
43
  AnyTool,
@@ -646,6 +650,7 @@ class TextEngine<
646
650
  this.deferredPromises.push(promise)
647
651
  },
648
652
  // Provider / adapter info
653
+ activity: 'chat',
649
654
  provider: config.adapter.name,
650
655
  model: config.params.model,
651
656
  source: 'server',
@@ -1197,6 +1202,23 @@ class TextEngine<
1197
1202
  }
1198
1203
  }
1199
1204
 
1205
+ /**
1206
+ * Tools available for execution this turn. The discovery tool is dropped
1207
+ * from the advertised set (`this.tools`) once every lazy tool is discovered,
1208
+ * but a model may still re-request discovery; this widens execution lookup
1209
+ * to include it so such calls don't fail with "Unknown tool". Centralised so
1210
+ * both execution sites (`processToolCalls` and `checkForPendingToolCalls`)
1211
+ * stay in sync.
1212
+ */
1213
+ private resolveExecutableTools(
1214
+ toolCalls: ReadonlyArray<ToolCall>,
1215
+ ): ReadonlyArray<AnyTool> {
1216
+ return this.lazyToolManager.getExecutableTools(
1217
+ this.tools,
1218
+ toolCalls.map((tc) => tc.function.name),
1219
+ )
1220
+ }
1221
+
1200
1222
  private async *checkForPendingToolCalls(): AsyncGenerator<
1201
1223
  StreamChunk,
1202
1224
  ToolPhaseResult,
@@ -1245,7 +1267,7 @@ class TextEngine<
1245
1267
 
1246
1268
  const generator = executeToolCalls(
1247
1269
  executablePendingCalls,
1248
- this.tools,
1270
+ this.resolveExecutableTools(executablePendingCalls),
1249
1271
  approvals,
1250
1272
  clientToolResults,
1251
1273
  (eventName, data) => this.createCustomEventChunk(eventName, data),
@@ -1407,7 +1429,7 @@ class TextEngine<
1407
1429
 
1408
1430
  const generator = executeToolCalls(
1409
1431
  executableToolCalls,
1410
- this.tools,
1432
+ this.resolveExecutableTools(executableToolCalls),
1411
1433
  approvals,
1412
1434
  clientToolResults,
1413
1435
  (eventName, data) => this.createCustomEventChunk(eventName, data),
@@ -2861,7 +2883,7 @@ async function* fallbackStructuredOutputStream(
2861
2883
  timestamp,
2862
2884
  }
2863
2885
 
2864
- let result: { data: unknown; rawText: string }
2886
+ let result: StructuredOutputResult<unknown>
2865
2887
  try {
2866
2888
  result = await adapter.structuredOutput(options)
2867
2889
  } catch (error) {
@@ -2917,6 +2939,12 @@ async function* fallbackStructuredOutputStream(
2917
2939
  model,
2918
2940
  timestamp,
2919
2941
  finishReason: 'stop',
2942
+ // Forward adapter-reported token usage so consumers reading
2943
+ // `RUN_FINISHED.usage` (and the engine's `runOnUsage` middleware hook) see
2944
+ // it on the fallback path, mirroring the native streaming path. The
2945
+ // conditional spread avoids emitting `usage: undefined` for adapters that
2946
+ // don't report it. See #758.
2947
+ ...(result.usage ? { usage: result.usage } : {}),
2920
2948
  }
2921
2949
  }
2922
2950
 
@@ -80,6 +80,13 @@ export interface ChatMiddlewareContext<TContext = unknown> {
80
80
 
81
81
  // --- Provider / adapter info (immutable for the lifetime of the request) ---
82
82
 
83
+ /**
84
+ * Which activity this context describes — always `'chat'`. Present so the
85
+ * chat context structurally satisfies the base `GenerationMiddlewareContext`,
86
+ * letting an observe-only middleware authored against the base (e.g.
87
+ * `otelMiddleware`) run on both chat and media activities.
88
+ */
89
+ activity: 'chat'
83
90
  /** Provider name (e.g., 'openai', 'anthropic') */
84
91
  provider: string
85
92
  /** Model identifier (e.g., 'gpt-4o') */
@@ -1,7 +1,14 @@
1
1
  import { convertSchemaToJsonSchema } from './schema-converter'
2
- import type { Tool } from '../../../types'
2
+ import type { AnyTool, Tool } from '../../../types'
3
3
 
4
- const DISCOVERY_TOOL_NAME = '__lazy__tool__discovery__'
4
+ /**
5
+ * Name of the synthetic tool the LLM calls to discover lazy tools.
6
+ *
7
+ * Exported so callers building custom message-compaction / history-trimming
8
+ * logic can reference the discovery tool by constant instead of hard-coding
9
+ * the string (which is an internal contract that could change).
10
+ */
11
+ export const DISCOVERY_TOOL_NAME = '__lazy__tool__discovery__'
5
12
 
6
13
  /**
7
14
  * Manages lazy tool discovery for the chat agent loop.
@@ -87,6 +94,35 @@ export class LazyToolManager {
87
94
  return active
88
95
  }
89
96
 
97
+ /**
98
+ * Returns the tools that should be available for *execution* this turn.
99
+ *
100
+ * This is the advertised set (`getActiveTools()`, passed in as `activeTools`)
101
+ * plus the discovery tool when a pending call references it but it is no
102
+ * longer advertised. Once every lazy tool has been discovered the discovery
103
+ * tool is dropped from the advertised set, but a model may still re-request
104
+ * discovery (long context / hallucination); keeping it executable lets that
105
+ * call return the schemas again instead of failing with "Unknown tool".
106
+ *
107
+ * The advertised set is intentionally left unchanged — only execution lookup
108
+ * is widened. Operates on the already-built `activeTools`: it must NOT call
109
+ * `getActiveTools()`, which would reset `hasNewDiscoveries` before the
110
+ * post-execution refresh check in the agent loop.
111
+ */
112
+ getExecutableTools(
113
+ activeTools: ReadonlyArray<AnyTool>,
114
+ pendingToolCallNames: ReadonlyArray<string>,
115
+ ): ReadonlyArray<AnyTool> {
116
+ if (
117
+ this.discoveryTool &&
118
+ pendingToolCallNames.includes(DISCOVERY_TOOL_NAME) &&
119
+ !activeTools.some((t) => t.name === DISCOVERY_TOOL_NAME)
120
+ ) {
121
+ return [...activeTools, this.discoveryTool]
122
+ }
123
+ return activeTools
124
+ }
125
+
90
126
  /**
91
127
  * Returns whether new tools have been discovered since the last getActiveTools() call.
92
128
  */
@@ -221,8 +257,14 @@ export class LazyToolManager {
221
257
  for (const name of args.toolNames) {
222
258
  const tool = lazyToolMap.get(name)
223
259
  if (tool) {
224
- manager.discoveredTools.add(name)
225
- manager.hasNewDiscoveries = true
260
+ // Only flag a refresh for genuinely new discoveries. Re-requesting
261
+ // an already-discovered tool still returns its schema below (the
262
+ // model asked for it), but must not trigger a redundant tool-list
263
+ // refresh + continue in the agent loop.
264
+ if (!manager.discoveredTools.has(name)) {
265
+ manager.discoveredTools.add(name)
266
+ manager.hasNewDiscoveries = true
267
+ }
226
268
  const jsonSchema = tool.inputSchema
227
269
  ? convertSchemaToJsonSchema(tool.inputSchema)
228
270
  : undefined
@@ -8,8 +8,16 @@
8
8
  import { aiEventClient } from '@tanstack/ai-event-client'
9
9
  import { streamGenerationResult } from '../stream-generation-result.js'
10
10
  import { resolveDebugOption } from '../../logger/resolve'
11
+ import {
12
+ createGenerationContext,
13
+ runGenerationError,
14
+ runGenerationFinish,
15
+ runGenerationStart,
16
+ runGenerationUsage,
17
+ } from '../middleware'
11
18
  import type { InternalLogger } from '../../logger/internal-logger'
12
19
  import type { DebugOption } from '../../logger/types'
20
+ import type { GenerationMiddleware } from '../middleware'
13
21
  import type { AudioAdapter } from './adapter'
14
22
  import type { AudioGenerationResult, StreamChunk } from '../../types'
15
23
 
@@ -70,6 +78,12 @@ export interface AudioActivityOptions<
70
78
  * control and/or a custom `Logger`.
71
79
  */
72
80
  debug?: DebugOption
81
+ /**
82
+ * Observe-only middleware notified on start, usage, success, and error. Pass
83
+ * `otelMiddleware()` to emit OpenTelemetry spans, or implement the
84
+ * `GenerationMiddleware` contract for a custom backend.
85
+ */
86
+ middleware?: Array<GenerationMiddleware>
73
87
  }
74
88
 
75
89
  // ===========================
@@ -135,7 +149,13 @@ async function runGenerateAudio<
135
149
  >(
136
150
  options: AudioActivityOptions<TAdapter, boolean>,
137
151
  ): Promise<AudioGenerationResult> {
138
- const { adapter, stream: _stream, debug: _debug, ...rest } = options
152
+ const {
153
+ adapter,
154
+ stream: _stream,
155
+ debug: _debug,
156
+ middleware,
157
+ ...rest
158
+ } = options
139
159
  const model = adapter.model
140
160
  const requestId = createId('audio')
141
161
  const startTime = Date.now()
@@ -145,6 +165,17 @@ async function runGenerateAudio<
145
165
  (adapter as { name?: string }).name ??
146
166
  'unknown'
147
167
 
168
+ const mwCtx = createGenerationContext({
169
+ requestId,
170
+ activity: 'audio',
171
+ provider: adapter.name,
172
+ model,
173
+ modelOptions: rest.modelOptions,
174
+ createId,
175
+ })
176
+
177
+ await runGenerationStart(middleware, mwCtx)
178
+
148
179
  aiEventClient.emit('audio:request:started', {
149
180
  requestId,
150
181
  provider: adapter.name,
@@ -189,6 +220,12 @@ async function runGenerateAudio<
189
220
  audioDuration: result.audio.duration,
190
221
  })
191
222
 
223
+ if (result.usage) await runGenerationUsage(middleware, mwCtx, result.usage)
224
+ await runGenerationFinish(middleware, mwCtx, {
225
+ duration: elapsedMs,
226
+ usage: result.usage,
227
+ })
228
+
192
229
  return result
193
230
  } catch (error) {
194
231
  const elapsedMs = Date.now() - startTime
@@ -202,6 +239,10 @@ async function runGenerateAudio<
202
239
  modelOptions: rest.modelOptions as Record<string, unknown> | undefined,
203
240
  timestamp: Date.now(),
204
241
  })
242
+ await runGenerationError(middleware, mwCtx, {
243
+ error,
244
+ duration: elapsedMs,
245
+ })
205
246
  logger.errors('generateAudio activity failed', {
206
247
  error,
207
248
  source: 'generateAudio',
@@ -1,4 +1,8 @@
1
- import type { ImageGenerationOptions, ImageGenerationResult } from '../../types'
1
+ import type {
2
+ ImageGenerationOptions,
3
+ ImageGenerationResult,
4
+ ModelInputModalitiesByName,
5
+ } from '../../types'
2
6
 
3
7
  /**
4
8
  * Resolve the size type for a model from the model-size map.
@@ -29,6 +33,8 @@ export interface ImageAdapterConfig {
29
33
  * - TProviderOptions: Base provider-specific options (already resolved)
30
34
  * - TModelProviderOptionsByName: Map from model name to its specific provider options
31
35
  * - TModelSizeByName: Map from model name to its supported sizes
36
+ * - TModelInputModalitiesByName: Map from model name to the non-text prompt
37
+ * modalities it accepts (constrains the `prompt` part types at compile time)
32
38
  */
33
39
  export interface ImageAdapter<
34
40
  TModel extends string = string,
@@ -38,6 +44,8 @@ export interface ImageAdapter<
38
44
  string,
39
45
  string
40
46
  >,
47
+ TModelInputModalitiesByName extends ModelInputModalitiesByName =
48
+ ModelInputModalitiesByName,
41
49
  > {
42
50
  /** Discriminator for adapter kind - used by generate() to determine API shape */
43
51
  readonly kind: 'image'
@@ -53,6 +61,7 @@ export interface ImageAdapter<
53
61
  providerOptions: TProviderOptions
54
62
  modelProviderOptionsByName: TModelProviderOptionsByName
55
63
  modelSizeByName: TModelSizeByName
64
+ modelInputModalitiesByName: TModelInputModalitiesByName
56
65
  }
57
66
 
58
67
  /**
@@ -67,7 +76,7 @@ export interface ImageAdapter<
67
76
  * An ImageAdapter with any/unknown type parameters.
68
77
  * Useful as a constraint in generic functions and interfaces.
69
78
  */
70
- export type AnyImageAdapter = ImageAdapter<any, any, any, any>
79
+ export type AnyImageAdapter = ImageAdapter<any, any, any, any, any>
71
80
 
72
81
  /**
73
82
  * Abstract base class for image generation adapters.
@@ -83,11 +92,14 @@ export abstract class BaseImageAdapter<
83
92
  string,
84
93
  string
85
94
  >,
95
+ TModelInputModalitiesByName extends ModelInputModalitiesByName =
96
+ ModelInputModalitiesByName,
86
97
  > implements ImageAdapter<
87
98
  TModel,
88
99
  TProviderOptions,
89
100
  TModelProviderOptionsByName,
90
- TModelSizeByName
101
+ TModelSizeByName,
102
+ TModelInputModalitiesByName
91
103
  > {
92
104
  readonly kind = 'image' as const
93
105
  abstract readonly name: string
@@ -98,6 +110,7 @@ export abstract class BaseImageAdapter<
98
110
  providerOptions: TProviderOptions
99
111
  modelProviderOptionsByName: TModelProviderOptionsByName
100
112
  modelSizeByName: TModelSizeByName
113
+ modelInputModalitiesByName: TModelInputModalitiesByName
101
114
  }
102
115
 
103
116
  protected config: ImageAdapterConfig