@tanstack/ai 0.54.0 → 0.55.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/index.js +17 -1
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/messages.d.ts +21 -1
- package/dist/esm/activities/chat/messages.js +50 -1
- package/dist/esm/activities/chat/messages.js.map +1 -1
- package/dist/esm/activities/chat/stream/processor.d.ts +17 -0
- package/dist/esm/activities/chat/stream/processor.js +27 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/package.json +3 -3
- package/skills/ai-core/adapter-configuration/SKILL.md +1 -1
- package/skills/ai-core/media-generation/SKILL.md +22 -12
- package/src/activities/chat/index.ts +23 -0
- package/src/activities/chat/messages.ts +60 -0
- package/src/activities/chat/stream/processor.ts +31 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.55.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -88,8 +88,8 @@
|
|
|
88
88
|
"@ag-ui/core": "0.1.1-canary.beta.0",
|
|
89
89
|
"@standard-schema/spec": "^1.1.0",
|
|
90
90
|
"partial-json": "^0.1.7",
|
|
91
|
-
"@tanstack/ai-
|
|
92
|
-
"@tanstack/ai-
|
|
91
|
+
"@tanstack/ai-utils": "^0.4.0",
|
|
92
|
+
"@tanstack/ai-event-client": "^0.11.3"
|
|
93
93
|
},
|
|
94
94
|
"peerDependencies": {
|
|
95
95
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -129,7 +129,7 @@ const adapterWithKey = createOpenaiChat('gpt-5.2', 'sk-...')
|
|
|
129
129
|
|
|
130
130
|
- `bedrockText(model)` or `bedrockText(model, { api: 'converse' })` (the default) — Bedrock's native Converse API via `@aws-sdk/client-bedrock-runtime` (adapter name `bedrock-converse`). Reaches the broad catalog: Claude, Nova, Llama, Mistral, DeepSeek, and more.
|
|
131
131
|
- `bedrockText(model, { api: 'chat' })` — OpenAI-compatible Chat Completions endpoint (adapter name `bedrock`). Open-weight models only (gpt-oss, DeepSeek V3.x, Gemma, Qwen, etc.). Does NOT reach Claude, Nova, or Llama.
|
|
132
|
-
- `bedrockText(model, { api: 'responses' })` — OpenAI-compatible Responses API, mantle-only (adapter name `bedrock-responses`). Currently gpt-oss
|
|
132
|
+
- `bedrockText(model, { api: 'responses' })` — OpenAI-compatible Responses API, mantle-only (adapter name `bedrock-responses`). Currently gpt-oss and Gemma 4.
|
|
133
133
|
|
|
134
134
|
Use `createBedrockText(model, apiKey, config?)` to pass the key explicitly. Auth resolves from `BEDROCK_API_KEY` / `AWS_BEARER_TOKEN_BEDROCK`, or SigV4 via the standard AWS credential chain (no extra packages needed — handled by `@aws-sdk/client-bedrock-runtime`).
|
|
135
135
|
|
|
@@ -5,7 +5,7 @@ description: >
|
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
|
|
6
6
|
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
7
|
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
|
|
8
|
-
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
8
|
+
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech/elevenlabsSpeech,
|
|
9
9
|
generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
|
|
10
10
|
useGenerateImage, useGenerateAudio,
|
|
11
11
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
@@ -20,6 +20,7 @@ sources:
|
|
|
20
20
|
- 'TanStack/ai:docs/media/audio-generation.md'
|
|
21
21
|
- 'TanStack/ai:docs/media/video-generation.md'
|
|
22
22
|
- 'TanStack/ai:docs/media/text-to-speech.md'
|
|
23
|
+
- 'TanStack/ai:docs/adapters/elevenlabs.md'
|
|
23
24
|
- 'TanStack/ai:docs/media/transcription.md'
|
|
24
25
|
- 'TanStack/ai:docs/advanced/debug-logging.md'
|
|
25
26
|
---
|
|
@@ -303,14 +304,14 @@ with `allowUrlFetch: true` on the adapter config
|
|
|
303
304
|
|
|
304
305
|
**Provider support matrix:**
|
|
305
306
|
|
|
306
|
-
| Provider | `generateImage` image parts | `generateVideo` image parts
|
|
307
|
-
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
308
|
-
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1.
|
|
309
|
-
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing).
|
|
310
|
-
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`.
|
|
311
|
-
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5).
|
|
312
|
-
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata.
|
|
313
|
-
| Anthropic | n/a (no image generation API). | n/a
|
|
307
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
308
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
309
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
310
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
|
|
311
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
312
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). On 1.5 a starting frame can be combined with reference inputs (it pins the first frame). A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
|
|
313
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
|
|
314
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
314
315
|
|
|
315
316
|
Video and audio prompt parts follow the same `metadata.role` convention
|
|
316
317
|
for video-to-video and lipsync flows on fal. Grok accepts one source
|
|
@@ -352,8 +353,14 @@ const { generate, result, isLoading } = useGenerateAudio({
|
|
|
352
353
|
|
|
353
354
|
### 3. Text-to-Speech
|
|
354
355
|
|
|
355
|
-
Adapters
|
|
356
|
-
`byteplusSpeech` (`seed-audio-1.0`).
|
|
356
|
+
Adapters include `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview),
|
|
357
|
+
`byteplusSpeech` (`seed-audio-1.0`), and `elevenlabsSpeech` (`eleven_v3`).
|
|
358
|
+
|
|
359
|
+
`elevenlabsSpeech` accepts `format: 'mp3' | 'pcm' | 'opus' | 'wav'`.
|
|
360
|
+
WAV output contains 44.1 kHz, 16-bit mono PCM with a RIFF header.
|
|
361
|
+
AAC and FLAC requests throw before the API call.
|
|
362
|
+
An explicit `modelOptions.outputFormat` overrides `format` and returns
|
|
363
|
+
the selected provider format without WAV wrapping.
|
|
357
364
|
|
|
358
365
|
> **BytePlus Seed Speech is a separate product from ModelArk** — it reads
|
|
359
366
|
> **`BYTEPLUS_VOICE_API_KEY`**, not `ARK_API_KEY`, and an Ark key there fails
|
|
@@ -364,7 +371,10 @@ Adapters: `openaiSpeech` (tts-1, tts-1-hd, gpt-4o-audio-preview) and
|
|
|
364
371
|
> drops `voice`. Voice ids ending `_uranus_bigtts` are TTS 2.0,
|
|
365
372
|
> `_mars_bigtts` / `_moon_bigtts` are TTS 1.0, and `*_emo_v2_*` are the 1.0
|
|
366
373
|
> voices that accept emotion tags. Formats: `wav`, `mp3`, `pcm`, `ogg_opus`;
|
|
367
|
-
> `watermark`
|
|
374
|
+
> `modelOptions.watermark` takes an object here, not a boolean:
|
|
375
|
+
> `{ aigc_watermark }` for an audible marker and `{ aigc_metadata: { enable } }`
|
|
376
|
+
> for header provenance. `watermark: true` is shorthand for
|
|
377
|
+
> `{ aigc_watermark: true }`.
|
|
368
378
|
|
|
369
379
|
```typescript
|
|
370
380
|
import { generateSpeech } from '@tanstack/ai'
|
|
@@ -63,10 +63,12 @@ import {
|
|
|
63
63
|
import { maxIterations as maxIterationsStrategy } from './agent-loop-strategies'
|
|
64
64
|
import { isCancelRequestedReason } from './cancel'
|
|
65
65
|
import {
|
|
66
|
+
appendUiResourceToModelMessages,
|
|
66
67
|
convertMessagesToModelMessages,
|
|
67
68
|
generateMessageId,
|
|
68
69
|
modelMessagesToUIMessages,
|
|
69
70
|
safeJsonStringify,
|
|
71
|
+
uiResourcePartFromCustomValue,
|
|
70
72
|
} from './messages'
|
|
71
73
|
import { MiddlewareRunner } from './middleware/compose'
|
|
72
74
|
import { getRunDetached } from './middleware/run-store'
|
|
@@ -2770,6 +2772,22 @@ class TextEngine<
|
|
|
2770
2772
|
}
|
|
2771
2773
|
}
|
|
2772
2774
|
|
|
2775
|
+
/**
|
|
2776
|
+
* Record a `ui-resource` CUSTOM chunk on the assistant ModelMessage owning
|
|
2777
|
+
* its `toolCallId` so the resource survives later MESSAGES_SNAPSHOT chunks
|
|
2778
|
+
* (e.g. the interrupt snapshot emitted when the run pauses on a client
|
|
2779
|
+
* tool). Mirrors the anchor-preserving approach used for
|
|
2780
|
+
* `toolCallMetadata` (#867). See #1397.
|
|
2781
|
+
*/
|
|
2782
|
+
private recordEmittedUiResource(value: unknown): void {
|
|
2783
|
+
const part = uiResourcePartFromCustomValue(value)
|
|
2784
|
+
if (!part) return
|
|
2785
|
+
const next = appendUiResourceToModelMessages(this.messages, part)
|
|
2786
|
+
if (next === this.messages) return
|
|
2787
|
+
this.messages = next
|
|
2788
|
+
this.middlewareCtx.messages = this.messages
|
|
2789
|
+
}
|
|
2790
|
+
|
|
2773
2791
|
private buildMessagesSnapshotChunk(): StreamChunk {
|
|
2774
2792
|
const withIds = this.messages.map((message, index) => ({
|
|
2775
2793
|
...message,
|
|
@@ -4431,6 +4449,11 @@ class TextEngine<
|
|
|
4431
4449
|
if (this.hasPublicRunStarted) continue
|
|
4432
4450
|
this.hasPublicRunStarted = true
|
|
4433
4451
|
}
|
|
4452
|
+
// Persist MCP Apps ui-resource emissions onto the tool-call anchor
|
|
4453
|
+
// message so interrupt MESSAGES_SNAPSHOT chunks keep them (#1397).
|
|
4454
|
+
if (spec.type === EventType.CUSTOM && spec.name === 'ui-resource') {
|
|
4455
|
+
this.recordEmittedUiResource((spec as CustomEvent).value)
|
|
4456
|
+
}
|
|
4434
4457
|
yield spec
|
|
4435
4458
|
this.middlewareCtx.chunkIndex++
|
|
4436
4459
|
}
|
|
@@ -387,6 +387,66 @@ function appendUiResources(
|
|
|
387
387
|
return { ...ui, parts: [...ui.parts, ...extra] }
|
|
388
388
|
}
|
|
389
389
|
|
|
390
|
+
/**
|
|
391
|
+
* Build a UIResourcePart from the value of a CUSTOM `ui-resource` chunk
|
|
392
|
+
* emitted via `ctx.emitCustomEvent('ui-resource', ...)` (MCP Apps). The
|
|
393
|
+
* emission-side value carries `resource`/`serverId`/`toolName` plus the
|
|
394
|
+
* `toolCallId` stamped by the tool-call context wrapper — the `type`
|
|
395
|
+
* discriminator is added here. Returns undefined when the value does not
|
|
396
|
+
* match the ui-resource shape.
|
|
397
|
+
*/
|
|
398
|
+
export function uiResourcePartFromCustomValue(
|
|
399
|
+
value: unknown,
|
|
400
|
+
): UIResourcePart | undefined {
|
|
401
|
+
if (!isRecord(value)) return undefined
|
|
402
|
+
const part: unknown = { type: 'ui-resource', ...value }
|
|
403
|
+
return isUiResourcePart(part) ? part : undefined
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* Store an emitted ui-resource part on the assistant ModelMessage that owns
|
|
408
|
+
* its `toolCallId` (the tool-call anchor), so it survives later
|
|
409
|
+
* MESSAGES_SNAPSHOT chunks — e.g. the interrupt snapshot emitted when the
|
|
410
|
+
* run pauses on a client tool (#1397). Mirrors how `toolCallMetadata` is
|
|
411
|
+
* preserved on the anchor (#867).
|
|
412
|
+
*
|
|
413
|
+
* Returns the SAME array reference when no anchor owns the tool call or the
|
|
414
|
+
* resource is already stored (idempotent).
|
|
415
|
+
*/
|
|
416
|
+
export function appendUiResourceToModelMessages(
|
|
417
|
+
messages: Array<ModelMessage>,
|
|
418
|
+
part: UIResourcePart,
|
|
419
|
+
): Array<ModelMessage> {
|
|
420
|
+
for (let index = messages.length - 1; index >= 0; index--) {
|
|
421
|
+
const message = messages[index]
|
|
422
|
+
if (!message || message.role !== 'assistant') continue
|
|
423
|
+
const ownsToolCall = message.toolCalls?.some(
|
|
424
|
+
(toolCall) => toolCall.id === part.toolCallId,
|
|
425
|
+
)
|
|
426
|
+
if (!ownsToolCall) continue
|
|
427
|
+
const previous = tanstackMetadata(message)?.uiResources ?? []
|
|
428
|
+
if (
|
|
429
|
+
previous.some((stored) => uiResourceKey(stored) === uiResourceKey(part))
|
|
430
|
+
) {
|
|
431
|
+
return messages
|
|
432
|
+
}
|
|
433
|
+
const nextMessage = {
|
|
434
|
+
...message,
|
|
435
|
+
metadata: {
|
|
436
|
+
...message.metadata,
|
|
437
|
+
tanstack: {
|
|
438
|
+
...tanstackMetadata(message),
|
|
439
|
+
uiResources: [...previous, part],
|
|
440
|
+
},
|
|
441
|
+
},
|
|
442
|
+
}
|
|
443
|
+
const next = messages.slice()
|
|
444
|
+
next[index] = nextMessage
|
|
445
|
+
return next
|
|
446
|
+
}
|
|
447
|
+
return messages
|
|
448
|
+
}
|
|
449
|
+
|
|
390
450
|
function assistantMetadata(
|
|
391
451
|
uiMessage: UIMessage,
|
|
392
452
|
): UIMessage['metadata'] | undefined {
|
|
@@ -242,6 +242,37 @@ export class StreamProcessor {
|
|
|
242
242
|
this.emitMessagesChange()
|
|
243
243
|
}
|
|
244
244
|
|
|
245
|
+
/**
|
|
246
|
+
* Put older UI messages at the front of the conversation.
|
|
247
|
+
*
|
|
248
|
+
* Skip a message if its id is already in the list. Keep the existing message.
|
|
249
|
+
* Then emit the same messages-change event as `setMessages`.
|
|
250
|
+
*
|
|
251
|
+
* Use this for older history pages. The first hydrate window uses `setMessages`.
|
|
252
|
+
*
|
|
253
|
+
* @param messages Older UI messages in insertion order. The first item is the oldest.
|
|
254
|
+
*
|
|
255
|
+
* @example
|
|
256
|
+
* ```ts
|
|
257
|
+
* processor.setMessages([newest])
|
|
258
|
+
* processor.prependMessages([oldest])
|
|
259
|
+
* ```
|
|
260
|
+
*/
|
|
261
|
+
prependMessages(messages: Array<UIMessage>) {
|
|
262
|
+
const existingIds = new Set(this.messages.map((message) => message.id))
|
|
263
|
+
const olderMessages: Array<UIMessage> = []
|
|
264
|
+
for (const message of messages) {
|
|
265
|
+
const isDuplicate = existingIds.has(message.id)
|
|
266
|
+
if (isDuplicate) {
|
|
267
|
+
continue
|
|
268
|
+
}
|
|
269
|
+
existingIds.add(message.id)
|
|
270
|
+
olderMessages.push(message)
|
|
271
|
+
}
|
|
272
|
+
this.messages = [...olderMessages, ...this.messages]
|
|
273
|
+
this.emitMessagesChange()
|
|
274
|
+
}
|
|
275
|
+
|
|
245
276
|
/**
|
|
246
277
|
* Add a user message to the conversation.
|
|
247
278
|
* Supports both simple string content and multimodal content arrays.
|