@tanstack/ai 0.44.0 → 0.44.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -4,9 +4,10 @@ description: >
|
|
|
4
4
|
Image, audio, video, speech (TTS), and transcription generation using
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
|
|
6
6
|
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
|
-
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo,
|
|
8
|
-
durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
9
|
-
with openaiTranscription/byteplusTranscription. React hooks:
|
|
7
|
+
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
|
|
8
|
+
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
9
|
+
generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
|
|
10
|
+
useGenerateImage, useGenerateAudio,
|
|
10
11
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
11
12
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
12
13
|
type: sub-skill
|
|
@@ -251,7 +252,8 @@ await generateImage({
|
|
|
251
252
|
],
|
|
252
253
|
})
|
|
253
254
|
|
|
254
|
-
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
|
|
255
|
+
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
|
|
256
|
+
// end_image_url; OpenRouter: frame_images + input_references)
|
|
255
257
|
import { generateVideo } from '@tanstack/ai'
|
|
256
258
|
import { falVideo } from '@tanstack/ai-fal'
|
|
257
259
|
|
|
@@ -282,25 +284,25 @@ with `allowUrlFetch: true` on the adapter config
|
|
|
282
284
|
|
|
283
285
|
**Role hints** (`metadata.role`):
|
|
284
286
|
|
|
285
|
-
| Role | Maps to
|
|
286
|
-
| --------------- |
|
|
287
|
-
| `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise
|
|
288
|
-
| `'character'` | Same as `'reference'`; Veo `referenceImages
|
|
289
|
-
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url`
|
|
290
|
-
| `'control'` | fal `control_image_url` (ControlNet / depth / pose)
|
|
291
|
-
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image`
|
|
292
|
-
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame`
|
|
287
|
+
| Role | Maps to |
|
|
288
|
+
| --------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
|
|
289
|
+
| `'reference'` | fal `reference_image_urls`; OpenRouter video `input_references[]`; Gemini multimodal part; positional otherwise |
|
|
290
|
+
| `'character'` | Same as `'reference'`; Veo `referenceImages`; OpenRouter `input_references[]` |
|
|
291
|
+
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
|
|
292
|
+
| `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
|
|
293
|
+
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); OpenRouter `frame_images[]` `first_frame`; Veo `image` |
|
|
294
|
+
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); OpenRouter `frame_images[]` `last_frame`; Veo `lastFrame` |
|
|
293
295
|
|
|
294
296
|
**Provider support matrix:**
|
|
295
297
|
|
|
296
|
-
| Provider | `generateImage` image parts | `generateVideo` image parts
|
|
297
|
-
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
298
|
-
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1.
|
|
299
|
-
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. |
|
|
300
|
-
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`.
|
|
301
|
-
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a
|
|
302
|
-
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. |
|
|
303
|
-
| Anthropic | n/a (no image generation API). | n/a
|
|
298
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
299
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
300
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
301
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
|
|
302
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
303
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
|
|
304
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
|
|
305
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
304
306
|
|
|
305
307
|
Video and audio prompt parts follow the same `metadata.role` convention
|
|
306
308
|
for video-to-video and lipsync flows on fal; other providers throw when
|
|
@@ -445,7 +447,13 @@ const { generate, result, isLoading } = useTranscription({
|
|
|
445
447
|
### 5. Video Generation (Experimental -- async polling)
|
|
446
448
|
|
|
447
449
|
Video generation uses a jobs/polling architecture. The server creates a job,
|
|
448
|
-
polls for status, and streams updates to the client.
|
|
450
|
+
polls for status, and streams updates to the client. Adapters: `openaiVideo`
|
|
451
|
+
(Sora), `geminiVideo` (Veo / Omni Flash), `grokVideo`, `byteplusVideo`
|
|
452
|
+
(Seedance), `falVideo` (Kling, MiniMax, Hunyuan, …), and `openRouterVideo`
|
|
453
|
+
(OpenRouter's dedicated `POST /api/v1/videos` gateway — Seedance, Veo, Wan,
|
|
454
|
+
Kling, Sora 2 Pro and others through one API key; `getVideoJobStatus()`
|
|
455
|
+
returns the video as a `data:` URL since OpenRouter's download URLs require
|
|
456
|
+
the API key, and surfaces the gateway-reported cost as `usage.cost`).
|
|
449
457
|
|
|
450
458
|
```typescript
|
|
451
459
|
import {
|
|
@@ -541,8 +549,9 @@ image-to-video only — needs an `image` prompt part as the starting frame, text
|
|
|
541
549
|
aspect-ratio size template like `'16:9_720p'`, integer durations 1-15s, reports
|
|
542
550
|
`usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
|
|
543
551
|
aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family,
|
|
544
|
-
4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
|
|
545
|
-
`
|
|
552
|
+
4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
|
|
553
|
+
`openRouterVideo(...)` (OpenRouter's dedicated `POST /api/v1/videos` gateway),
|
|
554
|
+
and `falVideo(...)` (hosted models, see cost tracking below).
|
|
546
555
|
|
|
547
556
|
> **Seedance option applicability is per model and enforced server-side** —
|
|
548
557
|
> Ark returns a 400 for an inapplicable field rather than ignoring it.
|
|
@@ -554,6 +563,29 @@ aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family
|
|
|
554
563
|
> days). Seedance is also reachable via `falVideo` — `byteplusVideo` is the
|
|
555
564
|
> direct-to-BytePlus path.
|
|
556
565
|
|
|
566
|
+
OpenRouter (`@tanstack/ai-openrouter`, `openRouterVideo`) runs the dedicated
|
|
567
|
+
async video API (`POST /api/v1/videos`) and shares the same typed-duration
|
|
568
|
+
contract — `duration`, `size`, and provider options are narrowed per model
|
|
569
|
+
from OpenRouter's published metadata, with the same `availableDurations()` /
|
|
570
|
+
`snapDuration()` helpers:
|
|
571
|
+
|
|
572
|
+
```typescript
|
|
573
|
+
import { openRouterVideo } from '@tanstack/ai-openrouter'
|
|
574
|
+
|
|
575
|
+
const adapter = openRouterVideo('bytedance/seedance-2.0')
|
|
576
|
+
adapter.availableDurations()
|
|
577
|
+
// { kind: 'discrete', values: [4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15] }
|
|
578
|
+
adapter.snapDuration(7.4) // 7
|
|
579
|
+
|
|
580
|
+
const sliderSeconds = 7 // raw seconds from a UI control
|
|
581
|
+
const { jobId } = await generateVideo({
|
|
582
|
+
adapter,
|
|
583
|
+
prompt: 'A timelapse of clouds',
|
|
584
|
+
duration: adapter.snapDuration(sliderSeconds),
|
|
585
|
+
})
|
|
586
|
+
// Completed url is a data: URL; usage.cost carries the real billed cost.
|
|
587
|
+
```
|
|
588
|
+
|
|
557
589
|
Client hook with job tracking:
|
|
558
590
|
|
|
559
591
|
```tsx
|
|
@@ -729,6 +729,8 @@ class TextEngine<
|
|
|
729
729
|
private streamStartTime = 0
|
|
730
730
|
private totalChunkCount = 0
|
|
731
731
|
private currentMessageId: string | null = null
|
|
732
|
+
private currentMessageCreatedAt: Date | null = null
|
|
733
|
+
private streamIdentityCaptured = false
|
|
732
734
|
private accumulatedContent = ''
|
|
733
735
|
private accumulatedThinking: Array<{ content: string; signature?: string }> =
|
|
734
736
|
[]
|
|
@@ -1268,6 +1270,8 @@ class TextEngine<
|
|
|
1268
1270
|
|
|
1269
1271
|
private async beginIteration(): Promise<void> {
|
|
1270
1272
|
this.currentMessageId = this.createId('msg')
|
|
1273
|
+
this.currentMessageCreatedAt = new Date()
|
|
1274
|
+
this.streamIdentityCaptured = false
|
|
1271
1275
|
this.accumulatedContent = ''
|
|
1272
1276
|
this.accumulatedThinking = []
|
|
1273
1277
|
this.currentThinkingContent = ''
|
|
@@ -1455,6 +1459,11 @@ class TextEngine<
|
|
|
1455
1459
|
// eslint-disable-next-line @typescript-eslint/switch-exhaustiveness-check -- AG-UI EventType enum members vs string-literal case labels; default branch handles untraced events.
|
|
1456
1460
|
switch (chunk.type) {
|
|
1457
1461
|
// AG-UI Events
|
|
1462
|
+
case 'TEXT_MESSAGE_START':
|
|
1463
|
+
if (typeof chunk.messageId === 'string' && chunk.messageId !== '') {
|
|
1464
|
+
this.captureStreamMessageIdentity(chunk.messageId)
|
|
1465
|
+
}
|
|
1466
|
+
break
|
|
1458
1467
|
case 'TEXT_MESSAGE_CONTENT':
|
|
1459
1468
|
this.handleTextMessageContentEvent(chunk)
|
|
1460
1469
|
break
|
|
@@ -1493,8 +1502,7 @@ class TextEngine<
|
|
|
1493
1502
|
break
|
|
1494
1503
|
|
|
1495
1504
|
default:
|
|
1496
|
-
// RUN_STARTED,
|
|
1497
|
-
// STATE_SNAPSHOT, STATE_DELTA, CUSTOM
|
|
1505
|
+
// RUN_STARTED, TEXT_MESSAGE_END, STATE_SNAPSHOT, STATE_DELTA, CUSTOM
|
|
1498
1506
|
// - no special handling needed in chat activity
|
|
1499
1507
|
break
|
|
1500
1508
|
}
|
|
@@ -1513,7 +1521,22 @@ class TextEngine<
|
|
|
1513
1521
|
this.middlewareCtx.accumulatedContent = this.accumulatedContent
|
|
1514
1522
|
}
|
|
1515
1523
|
|
|
1524
|
+
private captureStreamMessageIdentity(messageId: string): void {
|
|
1525
|
+
this.currentMessageId = messageId
|
|
1526
|
+
this.middlewareCtx.currentMessageId = messageId
|
|
1527
|
+
if (!this.streamIdentityCaptured) {
|
|
1528
|
+
this.currentMessageCreatedAt = new Date()
|
|
1529
|
+
this.streamIdentityCaptured = true
|
|
1530
|
+
}
|
|
1531
|
+
}
|
|
1532
|
+
|
|
1516
1533
|
private handleToolCallStartEvent(chunk: ToolCallStartEvent): void {
|
|
1534
|
+
if (
|
|
1535
|
+
typeof chunk.parentMessageId === 'string' &&
|
|
1536
|
+
chunk.parentMessageId !== ''
|
|
1537
|
+
) {
|
|
1538
|
+
this.captureStreamMessageIdentity(chunk.parentMessageId)
|
|
1539
|
+
}
|
|
1517
1540
|
this.toolCallManager.addToolCallStartEvent(chunk)
|
|
1518
1541
|
}
|
|
1519
1542
|
|
|
@@ -1954,6 +1977,8 @@ class TextEngine<
|
|
|
1954
1977
|
role: 'assistant',
|
|
1955
1978
|
content: this.accumulatedContent || null,
|
|
1956
1979
|
toolCalls,
|
|
1980
|
+
id: this.currentMessageId ?? undefined,
|
|
1981
|
+
createdAt: this.currentMessageCreatedAt ?? undefined,
|
|
1957
1982
|
...(this.accumulatedThinking.length > 0 && {
|
|
1958
1983
|
thinking: this.accumulatedThinking,
|
|
1959
1984
|
}),
|