@tanstack/ai 0.54.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.54.0",
3
+ "version": "0.55.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -88,8 +88,8 @@
88
88
  "@ag-ui/core": "0.1.1-canary.beta.0",
89
89
  "@standard-schema/spec": "^1.1.0",
90
90
  "partial-json": "^0.1.7",
91
- "@tanstack/ai-event-client": "^0.11.3",
92
- "@tanstack/ai-utils": "^0.4.0"
91
+ "@tanstack/ai-utils": "^0.4.0",
92
+ "@tanstack/ai-event-client": "^0.11.3"
93
93
  },
94
94
  "peerDependencies": {
95
95
  "@opentelemetry/api": ">=1.9.0"
@@ -129,7 +129,7 @@ const adapterWithKey = createOpenaiChat('gpt-5.2', 'sk-...')
129
129
 
130
130
  - `bedrockText(model)` or `bedrockText(model, { api: 'converse' })` (the default) — Bedrock's native Converse API via `@aws-sdk/client-bedrock-runtime` (adapter name `bedrock-converse`). Reaches the broad catalog: Claude, Nova, Llama, Mistral, DeepSeek, and more.
131
131
  - `bedrockText(model, { api: 'chat' })` — OpenAI-compatible Chat Completions endpoint (adapter name `bedrock`). Open-weight models only (gpt-oss, DeepSeek V3.x, Gemma, Qwen, etc.). Does NOT reach Claude, Nova, or Llama.
132
- - `bedrockText(model, { api: 'responses' })` — OpenAI-compatible Responses API, mantle-only (adapter name `bedrock-responses`). Currently gpt-oss family.
132
+ - `bedrockText(model, { api: 'responses' })` — OpenAI-compatible Responses API, mantle-only (adapter name `bedrock-responses`). Currently gpt-oss and Gemma 4.
133
133
 
134
134
  Use `createBedrockText(model, apiKey, config?)` to pass the key explicitly. Auth resolves from `BEDROCK_API_KEY` / `AWS_BEARER_TOKEN_BEDROCK`, or SigV4 via the standard AWS credential chain (no extra packages needed — handled by `@aws-sdk/client-bedrock-runtime`).
135
135
 
@@ -5,7 +5,7 @@ description: >
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
6
6
  generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
7
  polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
8
- per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
8
+ per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
9
9
  generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
10
10
  useGenerateImage, useGenerateAudio,
11
11
  useGenerateSpeech, useTranscription, useGenerateVideo.
@@ -20,6 +20,7 @@ sources:
20
20
  - 'TanStack/ai:docs/media/audio-generation.md'
21
21
  - 'TanStack/ai:docs/media/video-generation.md'
22
22
  - 'TanStack/ai:docs/media/text-to-speech.md'
23
+ - 'TanStack/ai:docs/adapters/elevenlabs.md'
23
24
  - 'TanStack/ai:docs/media/transcription.md'
24
25
  - 'TanStack/ai:docs/advanced/debug-logging.md'
25
26
  ---
@@ -303,14 +304,14 @@ with `allowUrlFetch: true` on the adapter config
303
304
 
304
305
  **Provider support matrix:**
305
306
 
306
- | Provider | `generateImage` image parts | `generateVideo` image parts |
307
- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
308
- | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
309
- | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
310
- | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
311
- | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). Starting frame and reference inputs cannot be combined. A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
312
- | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
313
- | Anthropic | n/a (no image generation API). | n/a |
307
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
308
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
309
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
310
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
311
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
312
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). On 1.5 a starting frame can be combined with reference inputs (it pins the first frame). A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
313
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
314
+ | Anthropic | n/a (no image generation API). | n/a |
314
315
 
315
316
  Video and audio prompt parts follow the same `metadata.role` convention
316
317
  for video-to-video and lipsync flows on fal. Grok accepts one source
@@ -352,8 +353,14 @@ const { generate, result, isLoading } = useGenerateAudio({
352
353
 
353
354
  ### 3. Text-to-Speech
354
355
 
355
- Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
356
- `byteplusSpeech` (`seed-audio-1.0`).
356
+ Adapters include `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview),
357
+ `byteplusSpeech` (`seed-audio-1.0`), and `elevenlabsSpeech` (`eleven_v3`).
358
+
359
+ `elevenlabsSpeech` accepts `format: 'mp3' | 'pcm' | 'opus' | 'wav'`.
360
+ WAV output contains 44.1 kHz, 16-bit mono PCM with a RIFF header.
361
+ AAC and FLAC requests throw before the API call.
362
+ An explicit `modelOptions.outputFormat` overrides `format` and returns
363
+ the selected provider format without WAV wrapping.
357
364
 
358
365
  > **BytePlus Seed Speech is a separate product from ModelArk** — it reads
359
366
  > **`BYTEPLUS_VOICE_API_KEY`**, not `ARK_API_KEY`, and an Ark key there fails
@@ -364,7 +371,10 @@ Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
364
371
  > drops `voice`. Voice ids ending `_uranus_bigtts` are TTS 2.0,
365
372
  > `_mars_bigtts` / `_moon_bigtts` are TTS 1.0, and `*_emo_v2_*` are the 1.0
366
373
  > voices that accept emotion tags. Formats: `wav`, `mp3`, `pcm`, `ogg_opus`;
367
- > `watermark` is also available on `modelOptions`.
374
+ > `modelOptions.watermark` takes an object here, not a boolean:
375
+ > `{ aigc_watermark }` for an audible marker and `{ aigc_metadata: { enable } }`
376
+ > for header provenance. `watermark: true` is shorthand for
377
+ > `{ aigc_watermark: true }`.
368
378
 
369
379
  ```typescript
370
380
  import { generateSpeech } from '@tanstack/ai'
@@ -63,10 +63,12 @@ import {
63
63
  import { maxIterations as maxIterationsStrategy } from './agent-loop-strategies'
64
64
  import { isCancelRequestedReason } from './cancel'
65
65
  import {
66
+ appendUiResourceToModelMessages,
66
67
  convertMessagesToModelMessages,
67
68
  generateMessageId,
68
69
  modelMessagesToUIMessages,
69
70
  safeJsonStringify,
71
+ uiResourcePartFromCustomValue,
70
72
  } from './messages'
71
73
  import { MiddlewareRunner } from './middleware/compose'
72
74
  import { getRunDetached } from './middleware/run-store'
@@ -2770,6 +2772,22 @@ class TextEngine<
2770
2772
  }
2771
2773
  }
2772
2774
 
2775
+ /**
2776
+ * Record a `ui-resource` CUSTOM chunk on the assistant ModelMessage owning
2777
+ * its `toolCallId` so the resource survives later MESSAGES_SNAPSHOT chunks
2778
+ * (e.g. the interrupt snapshot emitted when the run pauses on a client
2779
+ * tool). Mirrors the anchor-preserving approach used for
2780
+ * `toolCallMetadata` (#867). See #1397.
2781
+ */
2782
+ private recordEmittedUiResource(value: unknown): void {
2783
+ const part = uiResourcePartFromCustomValue(value)
2784
+ if (!part) return
2785
+ const next = appendUiResourceToModelMessages(this.messages, part)
2786
+ if (next === this.messages) return
2787
+ this.messages = next
2788
+ this.middlewareCtx.messages = this.messages
2789
+ }
2790
+
2773
2791
  private buildMessagesSnapshotChunk(): StreamChunk {
2774
2792
  const withIds = this.messages.map((message, index) => ({
2775
2793
  ...message,
@@ -4431,6 +4449,11 @@ class TextEngine<
4431
4449
  if (this.hasPublicRunStarted) continue
4432
4450
  this.hasPublicRunStarted = true
4433
4451
  }
4452
+ // Persist MCP Apps ui-resource emissions onto the tool-call anchor
4453
+ // message so interrupt MESSAGES_SNAPSHOT chunks keep them (#1397).
4454
+ if (spec.type === EventType.CUSTOM && spec.name === 'ui-resource') {
4455
+ this.recordEmittedUiResource((spec as CustomEvent).value)
4456
+ }
4434
4457
  yield spec
4435
4458
  this.middlewareCtx.chunkIndex++
4436
4459
  }
@@ -387,6 +387,66 @@ function appendUiResources(
387
387
  return { ...ui, parts: [...ui.parts, ...extra] }
388
388
  }
389
389
 
390
+ /**
391
+ * Build a UIResourcePart from the value of a CUSTOM `ui-resource` chunk
392
+ * emitted via `ctx.emitCustomEvent('ui-resource', ...)` (MCP Apps). The
393
+ * emission-side value carries `resource`/`serverId`/`toolName` plus the
394
+ * `toolCallId` stamped by the tool-call context wrapper — the `type`
395
+ * discriminator is added here. Returns undefined when the value does not
396
+ * match the ui-resource shape.
397
+ */
398
+ export function uiResourcePartFromCustomValue(
399
+ value: unknown,
400
+ ): UIResourcePart | undefined {
401
+ if (!isRecord(value)) return undefined
402
+ const part: unknown = { type: 'ui-resource', ...value }
403
+ return isUiResourcePart(part) ? part : undefined
404
+ }
405
+
406
+ /**
407
+ * Store an emitted ui-resource part on the assistant ModelMessage that owns
408
+ * its `toolCallId` (the tool-call anchor), so it survives later
409
+ * MESSAGES_SNAPSHOT chunks — e.g. the interrupt snapshot emitted when the
410
+ * run pauses on a client tool (#1397). Mirrors how `toolCallMetadata` is
411
+ * preserved on the anchor (#867).
412
+ *
413
+ * Returns the SAME array reference when no anchor owns the tool call or the
414
+ * resource is already stored (idempotent).
415
+ */
416
+ export function appendUiResourceToModelMessages(
417
+ messages: Array<ModelMessage>,
418
+ part: UIResourcePart,
419
+ ): Array<ModelMessage> {
420
+ for (let index = messages.length - 1; index >= 0; index--) {
421
+ const message = messages[index]
422
+ if (!message || message.role !== 'assistant') continue
423
+ const ownsToolCall = message.toolCalls?.some(
424
+ (toolCall) => toolCall.id === part.toolCallId,
425
+ )
426
+ if (!ownsToolCall) continue
427
+ const previous = tanstackMetadata(message)?.uiResources ?? []
428
+ if (
429
+ previous.some((stored) => uiResourceKey(stored) === uiResourceKey(part))
430
+ ) {
431
+ return messages
432
+ }
433
+ const nextMessage = {
434
+ ...message,
435
+ metadata: {
436
+ ...message.metadata,
437
+ tanstack: {
438
+ ...tanstackMetadata(message),
439
+ uiResources: [...previous, part],
440
+ },
441
+ },
442
+ }
443
+ const next = messages.slice()
444
+ next[index] = nextMessage
445
+ return next
446
+ }
447
+ return messages
448
+ }
449
+
390
450
  function assistantMetadata(
391
451
  uiMessage: UIMessage,
392
452
  ): UIMessage['metadata'] | undefined {
@@ -242,6 +242,37 @@ export class StreamProcessor {
242
242
  this.emitMessagesChange()
243
243
  }
244
244
 
245
+ /**
246
+ * Put older UI messages at the front of the conversation.
247
+ *
248
+ * Skip a message if its id is already in the list. Keep the existing message.
249
+ * Then emit the same messages-change event as `setMessages`.
250
+ *
251
+ * Use this for older history pages. The first hydrate window uses `setMessages`.
252
+ *
253
+ * @param messages Older UI messages in insertion order. The first item is the oldest.
254
+ *
255
+ * @example
256
+ * ```ts
257
+ * processor.setMessages([newest])
258
+ * processor.prependMessages([oldest])
259
+ * ```
260
+ */
261
+ prependMessages(messages: Array<UIMessage>) {
262
+ const existingIds = new Set(this.messages.map((message) => message.id))
263
+ const olderMessages: Array<UIMessage> = []
264
+ for (const message of messages) {
265
+ const isDuplicate = existingIds.has(message.id)
266
+ if (isDuplicate) {
267
+ continue
268
+ }
269
+ existingIds.add(message.id)
270
+ olderMessages.push(message)
271
+ }
272
+ this.messages = [...olderMessages, ...this.messages]
273
+ this.emitMessagesChange()
274
+ }
275
+
245
276
  /**
246
277
  * Add a user message to the conversation.
247
278
  * Supports both simple string content and multimodal content arrays.