@tanstack/ai 0.44.0 → 0.44.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.44.0",
3
+ "version": "0.44.1",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -4,9 +4,10 @@ description: >
4
4
  Image, audio, video, speech (TTS), and transcription generation using
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
6
6
  generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
- polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo, per-model typed
8
- durations), generateSpeech() with openaiSpeech/byteplusSpeech, generateTranscription()
9
- with openaiTranscription/byteplusTranscription. React hooks: useGenerateImage, useGenerateAudio,
7
+ polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
8
+ per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
9
+ generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
10
+ useGenerateImage, useGenerateAudio,
10
11
  useGenerateSpeech, useTranscription, useGenerateVideo.
11
12
  TanStack Start server function integration with toServerSentEventsResponse.
12
13
  type: sub-skill
@@ -251,7 +252,8 @@ await generateImage({
251
252
  ],
252
253
  })
253
254
 
254
- // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
255
+ // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
256
+ // end_image_url; OpenRouter: frame_images + input_references)
255
257
  import { generateVideo } from '@tanstack/ai'
256
258
  import { falVideo } from '@tanstack/ai-fal'
257
259
 
@@ -282,25 +284,25 @@ with `allowUrlFetch: true` on the adapter config
282
284
 
283
285
  **Role hints** (`metadata.role`):
284
286
 
285
- | Role | Maps to |
286
- | --------------- | ----------------------------------------------------------------------------------------------------- |
287
- | `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
288
- | `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
289
- | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
290
- | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
291
- | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
292
- | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
287
+ | Role | Maps to |
288
+ | --------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
289
+ | `'reference'` | fal `reference_image_urls`; OpenRouter video `input_references[]`; Gemini multimodal part; positional otherwise |
290
+ | `'character'` | Same as `'reference'`; Veo `referenceImages`; OpenRouter `input_references[]` |
291
+ | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
292
+ | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
293
+ | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); OpenRouter `frame_images[]` `first_frame`; Veo `image` |
294
+ | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); OpenRouter `frame_images[]` `last_frame`; Veo `lastFrame` |
293
295
 
294
296
  **Provider support matrix:**
295
297
 
296
- | Provider | `generateImage` image parts | `generateVideo` image parts |
297
- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
298
- | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
299
- | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
300
- | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
301
- | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
302
- | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
303
- | Anthropic | n/a (no image generation API). | n/a |
298
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
299
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
300
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
301
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
302
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
303
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
304
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
305
+ | Anthropic | n/a (no image generation API). | n/a |
304
306
 
305
307
  Video and audio prompt parts follow the same `metadata.role` convention
306
308
  for video-to-video and lipsync flows on fal; other providers throw when
@@ -445,7 +447,13 @@ const { generate, result, isLoading } = useTranscription({
445
447
  ### 5. Video Generation (Experimental -- async polling)
446
448
 
447
449
  Video generation uses a jobs/polling architecture. The server creates a job,
448
- polls for status, and streams updates to the client.
450
+ polls for status, and streams updates to the client. Adapters: `openaiVideo`
451
+ (Sora), `geminiVideo` (Veo / Omni Flash), `grokVideo`, `byteplusVideo`
452
+ (Seedance), `falVideo` (Kling, MiniMax, Hunyuan, …), and `openRouterVideo`
453
+ (OpenRouter's dedicated `POST /api/v1/videos` gateway — Seedance, Veo, Wan,
454
+ Kling, Sora 2 Pro and others through one API key; `getVideoJobStatus()`
455
+ returns the video as a `data:` URL since OpenRouter's download URLs require
456
+ the API key, and surfaces the gateway-reported cost as `usage.cost`).
449
457
 
450
458
  ```typescript
451
459
  import {
@@ -541,8 +549,9 @@ image-to-video only — needs an `image` prompt part as the starting frame, text
541
549
  aspect-ratio size template like `'16:9_720p'`, integer durations 1-15s, reports
542
550
  `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
543
551
  aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family,
544
- 4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`), and
545
- `falVideo(...)` (hosted models, see cost tracking below).
552
+ 4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
553
+ `openRouterVideo(...)` (OpenRouter's dedicated `POST /api/v1/videos` gateway),
554
+ and `falVideo(...)` (hosted models, see cost tracking below).
546
555
 
547
556
  > **Seedance option applicability is per model and enforced server-side** —
548
557
  > Ark returns a 400 for an inapplicable field rather than ignoring it.
@@ -554,6 +563,29 @@ aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family
554
563
  > days). Seedance is also reachable via `falVideo` — `byteplusVideo` is the
555
564
  > direct-to-BytePlus path.
556
565
 
566
+ OpenRouter (`@tanstack/ai-openrouter`, `openRouterVideo`) runs the dedicated
567
+ async video API (`POST /api/v1/videos`) and shares the same typed-duration
568
+ contract — `duration`, `size`, and provider options are narrowed per model
569
+ from OpenRouter's published metadata, with the same `availableDurations()` /
570
+ `snapDuration()` helpers:
571
+
572
+ ```typescript
573
+ import { openRouterVideo } from '@tanstack/ai-openrouter'
574
+
575
+ const adapter = openRouterVideo('bytedance/seedance-2.0')
576
+ adapter.availableDurations()
577
+ // { kind: 'discrete', values: [4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15] }
578
+ adapter.snapDuration(7.4) // 7
579
+
580
+ const sliderSeconds = 7 // raw seconds from a UI control
581
+ const { jobId } = await generateVideo({
582
+ adapter,
583
+ prompt: 'A timelapse of clouds',
584
+ duration: adapter.snapDuration(sliderSeconds),
585
+ })
586
+ // Completed url is a data: URL; usage.cost carries the real billed cost.
587
+ ```
588
+
557
589
  Client hook with job tracking:
558
590
 
559
591
  ```tsx
@@ -729,6 +729,8 @@ class TextEngine<
729
729
  private streamStartTime = 0
730
730
  private totalChunkCount = 0
731
731
  private currentMessageId: string | null = null
732
+ private currentMessageCreatedAt: Date | null = null
733
+ private streamIdentityCaptured = false
732
734
  private accumulatedContent = ''
733
735
  private accumulatedThinking: Array<{ content: string; signature?: string }> =
734
736
  []
@@ -1268,6 +1270,8 @@ class TextEngine<
1268
1270
 
1269
1271
  private async beginIteration(): Promise<void> {
1270
1272
  this.currentMessageId = this.createId('msg')
1273
+ this.currentMessageCreatedAt = new Date()
1274
+ this.streamIdentityCaptured = false
1271
1275
  this.accumulatedContent = ''
1272
1276
  this.accumulatedThinking = []
1273
1277
  this.currentThinkingContent = ''
@@ -1455,6 +1459,11 @@ class TextEngine<
1455
1459
  // eslint-disable-next-line @typescript-eslint/switch-exhaustiveness-check -- AG-UI EventType enum members vs string-literal case labels; default branch handles untraced events.
1456
1460
  switch (chunk.type) {
1457
1461
  // AG-UI Events
1462
+ case 'TEXT_MESSAGE_START':
1463
+ if (typeof chunk.messageId === 'string' && chunk.messageId !== '') {
1464
+ this.captureStreamMessageIdentity(chunk.messageId)
1465
+ }
1466
+ break
1458
1467
  case 'TEXT_MESSAGE_CONTENT':
1459
1468
  this.handleTextMessageContentEvent(chunk)
1460
1469
  break
@@ -1493,8 +1502,7 @@ class TextEngine<
1493
1502
  break
1494
1503
 
1495
1504
  default:
1496
- // RUN_STARTED, TEXT_MESSAGE_START, TEXT_MESSAGE_END,
1497
- // STATE_SNAPSHOT, STATE_DELTA, CUSTOM
1505
+ // RUN_STARTED, TEXT_MESSAGE_END, STATE_SNAPSHOT, STATE_DELTA, CUSTOM
1498
1506
  // - no special handling needed in chat activity
1499
1507
  break
1500
1508
  }
@@ -1513,7 +1521,22 @@ class TextEngine<
1513
1521
  this.middlewareCtx.accumulatedContent = this.accumulatedContent
1514
1522
  }
1515
1523
 
1524
+ private captureStreamMessageIdentity(messageId: string): void {
1525
+ this.currentMessageId = messageId
1526
+ this.middlewareCtx.currentMessageId = messageId
1527
+ if (!this.streamIdentityCaptured) {
1528
+ this.currentMessageCreatedAt = new Date()
1529
+ this.streamIdentityCaptured = true
1530
+ }
1531
+ }
1532
+
1516
1533
  private handleToolCallStartEvent(chunk: ToolCallStartEvent): void {
1534
+ if (
1535
+ typeof chunk.parentMessageId === 'string' &&
1536
+ chunk.parentMessageId !== ''
1537
+ ) {
1538
+ this.captureStreamMessageIdentity(chunk.parentMessageId)
1539
+ }
1517
1540
  this.toolCallManager.addToolCallStartEvent(chunk)
1518
1541
  }
1519
1542
 
@@ -1954,6 +1977,8 @@ class TextEngine<
1954
1977
  role: 'assistant',
1955
1978
  content: this.accumulatedContent || null,
1956
1979
  toolCalls,
1980
+ id: this.currentMessageId ?? undefined,
1981
+ createdAt: this.currentMessageCreatedAt ?? undefined,
1957
1982
  ...(this.accumulatedThinking.length > 0 && {
1958
1983
  thinking: this.accumulatedThinking,
1959
1984
  }),