@tanstack/ai 0.44.0 → 0.45.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +13 -1
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +126 -68
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/messages.js +24 -27
- package/dist/esm/activities/chat/messages.js.map +1 -1
- package/dist/esm/activities/chat/stream/processor.js +14 -13
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/chat/tools/approval-schema.js +11 -8
- package/dist/esm/activities/chat/tools/approval-schema.js.map +1 -1
- package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
- package/dist/esm/activities/chat/tools/schema-converter.js.map +1 -1
- package/dist/esm/activities/chat/tools/tool-calls.js +40 -31
- package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
- package/dist/esm/activities/generateVideo/index.js +3 -2
- package/dist/esm/activities/generateVideo/index.js.map +1 -1
- package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
- package/dist/esm/adapter-internals.d.ts +2 -0
- package/dist/esm/adapter-internals.js +3 -1
- package/dist/esm/interrupt-resume.js +24 -20
- package/dist/esm/interrupt-resume.js.map +1 -1
- package/dist/esm/interrupts.js +2 -1
- package/dist/esm/interrupts.js.map +1 -1
- package/dist/esm/logger/console-logger.js +1 -3
- package/dist/esm/logger/console-logger.js.map +1 -1
- package/dist/esm/logger/resolve.js +2 -1
- package/dist/esm/logger/resolve.js.map +1 -1
- package/dist/esm/stream-durability.js +1 -1
- package/dist/esm/stream-durability.js.map +1 -1
- package/dist/esm/types.d.ts +19 -16
- package/dist/esm/utilities/chat-params.js +1 -3
- package/dist/esm/utilities/chat-params.js.map +1 -1
- package/dist/esm/utilities/media-prompt.js +1 -3
- package/dist/esm/utilities/media-prompt.js.map +1 -1
- package/dist/esm/utilities/structured-output-events.d.ts +17 -0
- package/dist/esm/utilities/structured-output-events.js +32 -0
- package/dist/esm/utilities/structured-output-events.js.map +1 -0
- package/dist/esm/utilities/structured-output-text.d.ts +7 -0
- package/dist/esm/utilities/structured-output-text.js +50 -0
- package/dist/esm/utilities/structured-output-text.js.map +1 -0
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +6 -2
- package/skills/ai-core/media-generation/SKILL.md +80 -34
- package/skills/ai-core/structured-outputs/SKILL.md +73 -1
- package/skills/ai-core/tool-calling/SKILL.md +1 -1
- package/src/activities/chat/adapter.ts +16 -1
- package/src/activities/chat/index.ts +155 -49
- package/src/activities/chat/stream/processor.ts +6 -3
- package/src/activities/chat/tools/tool-calls.ts +12 -4
- package/src/adapter-internals.ts +8 -0
- package/src/types.ts +27 -19
- package/src/utilities/structured-output-events.ts +44 -0
- package/src/utilities/structured-output-text.ts +63 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.45.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -93,7 +93,7 @@
|
|
|
93
93
|
},
|
|
94
94
|
"devDependencies": {
|
|
95
95
|
"@opentelemetry/api": "^1.9.0",
|
|
96
|
-
"@vitest/coverage-v8": "4.
|
|
96
|
+
"@vitest/coverage-v8": "4.1.10",
|
|
97
97
|
"arktype": "^2.1.28",
|
|
98
98
|
"zod": "^4.2.0"
|
|
99
99
|
},
|
|
@@ -94,8 +94,12 @@ Note: `GOOGLE_GENAI_API_KEY` does NOT work.
|
|
|
94
94
|
## Gotchas
|
|
95
95
|
|
|
96
96
|
- All Gemini models are multimodal (text, image, audio, video, document input).
|
|
97
|
-
- Image generation models (`gemini-3-pro-image-
|
|
98
|
-
input limits (65K tokens) compared to text models (1M tokens).
|
|
97
|
+
- Image generation models (`gemini-3-pro-image`, `gemini-3.1-flash-image`, etc.)
|
|
98
|
+
have smaller input limits (65K tokens) compared to text models (1M tokens).
|
|
99
|
+
- Use the GA image ids. `gemini-3-pro-image-preview` and
|
|
100
|
+
`gemini-3.1-flash-image-preview` were shut down on 2026-06-25 and now 404;
|
|
101
|
+
they remain in the type union only as deprecated aliases, so a call to them
|
|
102
|
+
compiles and then fails at runtime.
|
|
99
103
|
- `thinkingConfig.thinkingLevel` (level-based) and `thinkingConfig.thinkingBudget`
|
|
100
104
|
(budget-based) serve different models. Check which your model supports.
|
|
101
105
|
- `cachedContent` must follow the format `cachedContents/{id}`.
|
|
@@ -4,9 +4,10 @@ description: >
|
|
|
4
4
|
Image, audio, video, speech (TTS), and transcription generation using
|
|
5
5
|
activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
|
|
6
6
|
generateAudio() with geminiAudio/falAudio, generateVideo() with async
|
|
7
|
-
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo,
|
|
8
|
-
durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
9
|
-
with openaiTranscription/byteplusTranscription. React hooks:
|
|
7
|
+
polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
|
|
8
|
+
per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
|
|
9
|
+
generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
|
|
10
|
+
useGenerateImage, useGenerateAudio,
|
|
10
11
|
useGenerateSpeech, useTranscription, useGenerateVideo.
|
|
11
12
|
TanStack Start server function integration with toServerSentEventsResponse.
|
|
12
13
|
type: sub-skill
|
|
@@ -150,9 +151,16 @@ function ImageGenerator() {
|
|
|
150
151
|
### 1. Image Generation
|
|
151
152
|
|
|
152
153
|
Supported adapters: `openaiImage` (dall-e-2, dall-e-3, gpt-image-1,
|
|
153
|
-
gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image
|
|
154
|
-
gemini-3.1-flash-lite-image, imagen-4.0-generate-001, etc.)
|
|
155
|
-
(Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`,
|
|
154
|
+
gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image,
|
|
155
|
+
gemini-3.1-flash-lite-image, gemini-3-pro-image, imagen-4.0-generate-001, etc.)
|
|
156
|
+
and `byteplusImage` (Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`,
|
|
157
|
+
the 5.0 family).
|
|
158
|
+
|
|
159
|
+
> **Use the GA Gemini image ids.** `gemini-3.1-flash-image-preview` and
|
|
160
|
+
> `gemini-3-pro-image-preview` were shut down on 2026-06-25 and now 404. They
|
|
161
|
+
> survive in the type union only as deprecated aliases so existing code keeps
|
|
162
|
+
> compiling — a call to them typechecks and then fails at runtime. Use
|
|
163
|
+
> `gemini-3.1-flash-image` / `gemini-3-pro-image` instead.
|
|
156
164
|
|
|
157
165
|
> **Seedream quirks:** `watermark` defaults to **`true`** (pass
|
|
158
166
|
> `modelOptions: { watermark: false }` for a clean image), `size` is a token
|
|
@@ -181,7 +189,7 @@ const openaiResult = await generateImage({
|
|
|
181
189
|
|
|
182
190
|
// Gemini native model with aspect-ratio sizes
|
|
183
191
|
const geminiResult = await generateImage({
|
|
184
|
-
adapter: geminiImage('gemini-3.1-flash-image
|
|
192
|
+
adapter: geminiImage('gemini-3.1-flash-image'),
|
|
185
193
|
prompt: 'A futuristic cityscape at night',
|
|
186
194
|
size: '16:9_4K',
|
|
187
195
|
})
|
|
@@ -251,7 +259,8 @@ await generateImage({
|
|
|
251
259
|
],
|
|
252
260
|
})
|
|
253
261
|
|
|
254
|
-
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
|
|
262
|
+
// Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
|
|
263
|
+
// end_image_url; OpenRouter: frame_images + input_references)
|
|
255
264
|
import { generateVideo } from '@tanstack/ai'
|
|
256
265
|
import { falVideo } from '@tanstack/ai-fal'
|
|
257
266
|
|
|
@@ -282,29 +291,30 @@ with `allowUrlFetch: true` on the adapter config
|
|
|
282
291
|
|
|
283
292
|
**Role hints** (`metadata.role`):
|
|
284
293
|
|
|
285
|
-
| Role | Maps to
|
|
286
|
-
| --------------- |
|
|
287
|
-
| `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise
|
|
288
|
-
| `'character'` | Same as `'reference'`; Veo `referenceImages
|
|
289
|
-
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url`
|
|
290
|
-
| `'control'` | fal `control_image_url` (ControlNet / depth / pose)
|
|
291
|
-
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image`
|
|
292
|
-
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame`
|
|
294
|
+
| Role | Maps to |
|
|
295
|
+
| --------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
|
|
296
|
+
| `'reference'` | fal `reference_image_urls`; OpenRouter video `input_references[]`; Gemini multimodal part; positional otherwise |
|
|
297
|
+
| `'character'` | Same as `'reference'`; Veo `referenceImages`; OpenRouter `input_references[]` |
|
|
298
|
+
| `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
|
|
299
|
+
| `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
|
|
300
|
+
| `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); OpenRouter `frame_images[]` `first_frame`; Veo `image` |
|
|
301
|
+
| `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); OpenRouter `frame_images[]` `last_frame`; Veo `lastFrame` |
|
|
293
302
|
|
|
294
303
|
**Provider support matrix:**
|
|
295
304
|
|
|
296
|
-
| Provider | `generateImage` image parts | `generateVideo` image parts
|
|
297
|
-
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
298
|
-
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1.
|
|
299
|
-
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. |
|
|
300
|
-
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`.
|
|
301
|
-
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. |
|
|
302
|
-
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. |
|
|
303
|
-
| Anthropic | n/a (no image generation API). | n/a
|
|
305
|
+
| Provider | `generateImage` image parts | `generateVideo` image parts |
|
|
306
|
+
| ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
307
|
+
| OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
|
|
308
|
+
| Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
|
|
309
|
+
| fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
|
|
310
|
+
| Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). Starting frame and reference inputs cannot be combined. A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
|
|
311
|
+
| OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
|
|
312
|
+
| Anthropic | n/a (no image generation API). | n/a |
|
|
304
313
|
|
|
305
314
|
Video and audio prompt parts follow the same `metadata.role` convention
|
|
306
|
-
for video-to-video and lipsync flows on fal
|
|
307
|
-
|
|
315
|
+
for video-to-video and lipsync flows on fal. Grok accepts one source
|
|
316
|
+
`video` part on `grok-imagine-video` with `modelOptions.mode: 'edit' | 'extend'`
|
|
317
|
+
and rejects audio parts. Other providers throw when those parts are passed.
|
|
308
318
|
|
|
309
319
|
### 2. Audio Generation (Music, Sound Effects)
|
|
310
320
|
|
|
@@ -445,7 +455,13 @@ const { generate, result, isLoading } = useTranscription({
|
|
|
445
455
|
### 5. Video Generation (Experimental -- async polling)
|
|
446
456
|
|
|
447
457
|
Video generation uses a jobs/polling architecture. The server creates a job,
|
|
448
|
-
polls for status, and streams updates to the client.
|
|
458
|
+
polls for status, and streams updates to the client. Adapters: `openaiVideo`
|
|
459
|
+
(Sora), `geminiVideo` (Veo / Omni Flash), `grokVideo`, `byteplusVideo`
|
|
460
|
+
(Seedance), `falVideo` (Kling, MiniMax, Hunyuan, …), and `openRouterVideo`
|
|
461
|
+
(OpenRouter's dedicated `POST /api/v1/videos` gateway — Seedance, Veo, Wan,
|
|
462
|
+
Kling, Sora 2 Pro and others through one API key; `getVideoJobStatus()`
|
|
463
|
+
returns the video as a `data:` URL since OpenRouter's download URLs require
|
|
464
|
+
the API key, and surfaces the gateway-reported cost as `usage.cost`).
|
|
449
465
|
|
|
450
466
|
```typescript
|
|
451
467
|
import {
|
|
@@ -536,13 +552,20 @@ const edited = await generateVideo({
|
|
|
536
552
|
|
|
537
553
|
Other video adapters: `openaiVideo('sora-2')` (pixel sizes like `'1280x720'`,
|
|
538
554
|
durations 4/8/12s, single `input_reference` image prompt part), `grokVideo(...)`
|
|
539
|
-
(`grok-imagine-video`
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
555
|
+
(`grok-imagine-video` and `grok-imagine-video-1.5` both do text-to-video + image-to-video;
|
|
556
|
+
1.5 adds reference-to-video — `'reference'`/`'character'`-roled image parts →
|
|
557
|
+
`reference_images` (max 7), preset voices via `modelOptions.reference_audios` (max 3) —
|
|
558
|
+
1.5-only, capped at 720p, and not combinable with a starting-frame image; only
|
|
559
|
+
`grok-imagine-video` edits/extends a source `video` prompt part via
|
|
560
|
+
`modelOptions.mode: 'edit' | 'extend'` (extend `duration` = added tail). Edit/extend
|
|
561
|
+
outputs inherit the source clip's properties, so `size`/`aspect_ratio`/`resolution`
|
|
562
|
+
throw in both modes and `duration` throws in edit mode — pass none of them there;
|
|
563
|
+
generation uses the aspect-ratio size template like `'16:9_720p'` (1080p is 1.5-only),
|
|
564
|
+
integer durations 1-15s, reports `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
|
|
543
565
|
aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family,
|
|
544
|
-
4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
|
|
545
|
-
`
|
|
566
|
+
4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
|
|
567
|
+
`openRouterVideo(...)` (OpenRouter's dedicated `POST /api/v1/videos` gateway),
|
|
568
|
+
and `falVideo(...)` (hosted models, see cost tracking below).
|
|
546
569
|
|
|
547
570
|
> **Seedance option applicability is per model and enforced server-side** —
|
|
548
571
|
> Ark returns a 400 for an inapplicable field rather than ignoring it.
|
|
@@ -554,6 +577,29 @@ aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family
|
|
|
554
577
|
> days). Seedance is also reachable via `falVideo` — `byteplusVideo` is the
|
|
555
578
|
> direct-to-BytePlus path.
|
|
556
579
|
|
|
580
|
+
OpenRouter (`@tanstack/ai-openrouter`, `openRouterVideo`) runs the dedicated
|
|
581
|
+
async video API (`POST /api/v1/videos`) and shares the same typed-duration
|
|
582
|
+
contract — `duration`, `size`, and provider options are narrowed per model
|
|
583
|
+
from OpenRouter's published metadata, with the same `availableDurations()` /
|
|
584
|
+
`snapDuration()` helpers:
|
|
585
|
+
|
|
586
|
+
```typescript
|
|
587
|
+
import { openRouterVideo } from '@tanstack/ai-openrouter'
|
|
588
|
+
|
|
589
|
+
const adapter = openRouterVideo('bytedance/seedance-2.0')
|
|
590
|
+
adapter.availableDurations()
|
|
591
|
+
// { kind: 'discrete', values: [4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15] }
|
|
592
|
+
adapter.snapDuration(7.4) // 7
|
|
593
|
+
|
|
594
|
+
const sliderSeconds = 7 // raw seconds from a UI control
|
|
595
|
+
const { jobId } = await generateVideo({
|
|
596
|
+
adapter,
|
|
597
|
+
prompt: 'A timelapse of clouds',
|
|
598
|
+
duration: adapter.snapDuration(sliderSeconds),
|
|
599
|
+
})
|
|
600
|
+
// Completed url is a data: URL; usage.cost carries the real billed cost.
|
|
601
|
+
```
|
|
602
|
+
|
|
557
603
|
Client hook with job tracking:
|
|
558
604
|
|
|
559
605
|
```tsx
|
|
@@ -964,7 +1010,7 @@ generateImage({
|
|
|
964
1010
|
})
|
|
965
1011
|
|
|
966
1012
|
generateImage({
|
|
967
|
-
adapter: geminiImage('gemini-3.1-flash-image
|
|
1013
|
+
adapter: geminiImage('gemini-3.1-flash-image'), // native multimodal
|
|
968
1014
|
prompt: [
|
|
969
1015
|
{ type: 'text', content: 'Edit this' },
|
|
970
1016
|
{ type: 'image', source: { type: 'url', value: url } },
|
|
@@ -20,6 +20,7 @@ sources:
|
|
|
20
20
|
- 'TanStack/ai:docs/structured-outputs/streaming.md'
|
|
21
21
|
- 'TanStack/ai:docs/structured-outputs/multi-turn.md'
|
|
22
22
|
- 'TanStack/ai:docs/structured-outputs/with-tools.md'
|
|
23
|
+
- 'TanStack/ai:docs/structured-outputs/harnesses.md'
|
|
23
24
|
---
|
|
24
25
|
|
|
25
26
|
# Structured Outputs
|
|
@@ -48,7 +49,7 @@ person.age // number
|
|
|
48
49
|
|
|
49
50
|
When `outputSchema` is provided, `chat()` returns `Promise<InferSchemaType<TSchema>>` instead of `AsyncIterable<StreamChunk>`. The result is fully typed.
|
|
50
51
|
|
|
51
|
-
Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client,
|
|
52
|
+
Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client, **Pattern 5** for multi-turn structured chats, and **Pattern 6** for harness adapters.
|
|
52
53
|
|
|
53
54
|
## Decision: which pattern fits
|
|
54
55
|
|
|
@@ -59,6 +60,7 @@ Adding `stream: true` switches the return to `StructuredOutputStream<InferSchema
|
|
|
59
60
|
| Direct iteration of the stream in Node or tests | Pattern 3 — async iterable |
|
|
60
61
|
| Users iterate on a structured object across multiple turns (recipe builder, ticket refinement) | Pattern 5 — multi-turn structured chat |
|
|
61
62
|
| Tools that gather info, then return a typed object | Combine any of the above with `tools` — see ai-core/tool-calling |
|
|
63
|
+
| A coding agent in a sandbox inspects files, then returns a typed object | Pattern 6 — harness `outputSchema` |
|
|
62
64
|
|
|
63
65
|
## Core Patterns
|
|
64
66
|
|
|
@@ -189,6 +191,13 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
|
|
|
189
191
|
| `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
|
|
190
192
|
| `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
|
|
191
193
|
| `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
|
|
194
|
+
| `@tanstack/ai-bedrock` | Separate native `structuredOutputStream` finalization through Converse or an OpenAI-compatible API |
|
|
195
|
+
| `@tanstack/ai-byteplus` | Native combined mode on supported models; unsupported models emit `RUN_ERROR` |
|
|
196
|
+
| `@tanstack/ai-claude-code` | Combined + event source — `--json-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
|
|
197
|
+
| `@tanstack/ai-codex` | Combined + event source — `--output-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
|
|
198
|
+
| `@tanstack/ai-opencode` | Combined + event source — prompt-and-parse. Read `useChat().final`. See Pattern 6. |
|
|
199
|
+
| `@tanstack/ai-grok-build` | Combined + event source — prompt-and-parse (ACP and streaming-json). Read `useChat().final` or the `structured-output` part. See Pattern 6. |
|
|
200
|
+
| `@tanstack/ai-acp` (`acpCompatible`) | Combined + event source — prompt-and-parse. Read `useChat().final` or the `structured-output` part. See Pattern 6. |
|
|
192
201
|
| All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
|
|
193
202
|
|
|
194
203
|
**Native combined mode vs fallback** is signaled by the adapter's
|
|
@@ -341,6 +350,68 @@ Key behaviors:
|
|
|
341
350
|
- **`partial` / `final` are derived.** The hook-level `partial` and `final` are NOT singleton state — they're derived from the latest assistant message's part (the one after the most recent user message). Between `sendMessage()` and the first chunk, `partial` reads `{}` and `final` reads `null` because no new assistant turn exists yet.
|
|
342
351
|
- **Round-trip preserves history.** When the client sends turn N+1, each prior assistant turn's `structured-output` part is serialized back as `{ role: 'assistant', content: <part.raw> }` so the model sees its own prior structured response. Streaming / errored parts are dropped from the round-trip.
|
|
343
352
|
|
|
353
|
+
### Pattern 6: Harness adapters (Claude Code, Codex, OpenCode, Grok Build, ACP)
|
|
354
|
+
|
|
355
|
+
Dedicated harness adapters honor `chat({ outputSchema })` on the same turn. Native harness tools still run. Read the object from `await chat()`, from `useChat().final`, or from the assistant `structured-output` part on `messages[].parts`. Do not parse assistant prose.
|
|
356
|
+
|
|
357
|
+
A UI endpoint must pass `stream: true`. Without it, `chat()` returns a `Promise`, not SSE.
|
|
358
|
+
|
|
359
|
+
```typescript
|
|
360
|
+
import { chat, toServerSentEventsResponse } from '@tanstack/ai'
|
|
361
|
+
import { claudeCodeText } from '@tanstack/ai-claude-code'
|
|
362
|
+
import { withSandbox } from '@tanstack/ai-sandbox'
|
|
363
|
+
import { z } from 'zod'
|
|
364
|
+
import { sandbox } from './sandbox'
|
|
365
|
+
|
|
366
|
+
const ReportSchema = z.object({
|
|
367
|
+
name: z.string(),
|
|
368
|
+
oneLiner: z.string(),
|
|
369
|
+
})
|
|
370
|
+
|
|
371
|
+
export async function POST(request: Request) {
|
|
372
|
+
const body: unknown = await request.json()
|
|
373
|
+
const messages =
|
|
374
|
+
typeof body === 'object' &&
|
|
375
|
+
body !== null &&
|
|
376
|
+
'messages' in body &&
|
|
377
|
+
Array.isArray(body.messages)
|
|
378
|
+
? body.messages
|
|
379
|
+
: []
|
|
380
|
+
|
|
381
|
+
const stream = chat({
|
|
382
|
+
adapter: claudeCodeText('claude-opus-4-8'),
|
|
383
|
+
messages,
|
|
384
|
+
outputSchema: ReportSchema,
|
|
385
|
+
stream: true,
|
|
386
|
+
middleware: [withSandbox(sandbox)],
|
|
387
|
+
})
|
|
388
|
+
return toServerSentEventsResponse(stream)
|
|
389
|
+
}
|
|
390
|
+
```
|
|
391
|
+
|
|
392
|
+
```tsx
|
|
393
|
+
import { useChat, fetchServerSentEvents } from '@tanstack/ai-react'
|
|
394
|
+
import { z } from 'zod'
|
|
395
|
+
|
|
396
|
+
const ReportSchema = z.object({
|
|
397
|
+
name: z.string(),
|
|
398
|
+
oneLiner: z.string(),
|
|
399
|
+
})
|
|
400
|
+
|
|
401
|
+
const { final } = useChat({
|
|
402
|
+
connection: fetchServerSentEvents('/api/repo-report'),
|
|
403
|
+
outputSchema: ReportSchema,
|
|
404
|
+
})
|
|
405
|
+
|
|
406
|
+
final?.name
|
|
407
|
+
```
|
|
408
|
+
|
|
409
|
+
- Claude Code: `--json-schema`. Codex: `--output-schema`. OpenCode, Grok Build, and `acpCompatible`: prompt-and-parse.
|
|
410
|
+
- `partial` stays empty until `structured-output.complete`.
|
|
411
|
+
- Client tools and `needsApproval` fail fast. The harness cannot pause for a browser round-trip.
|
|
412
|
+
- Render live work from `messages[].parts` (`thinking`, `tool-call`, `text`, `structured-output`). `final` is only the latest turn.
|
|
413
|
+
- See [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md).
|
|
414
|
+
|
|
344
415
|
## Common Mistakes
|
|
345
416
|
|
|
346
417
|
### HIGH: Filtering `TextPart`s out of `useChat` renderers when using `outputSchema`
|
|
@@ -509,4 +580,5 @@ provider call, stripping system prompts), use the dedicated
|
|
|
509
580
|
- See also: **ai-core/chat-experience/SKILL.md** — Base `useChat` surface; the structured-output additions documented here layer on top.
|
|
510
581
|
- See also: **ai-core/adapter-configuration/SKILL.md** — Adapter handles structured-output strategy transparently.
|
|
511
582
|
- See also: **ai-core/tool-calling/SKILL.md** — Combine `tools` with `outputSchema` for an agent loop that runs tools first and returns a typed object. Tool-approval and client-tool flows compose with structured runs without extra wiring; see [docs/structured-outputs/with-tools.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/with-tools.md).
|
|
583
|
+
- See also: [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md) — dedicated harness adapters and `useChat().final`.
|
|
512
584
|
- See also: **ai-core/middleware/SKILL.md** — `onStructuredOutputConfig` hook and the `structuredOutput` phase for observing/transforming the final structured-output call.
|
|
@@ -623,7 +623,7 @@ export const Route = createFileRoute('/api/chat')({
|
|
|
623
623
|
|
|
624
624
|
## Provider Skills
|
|
625
625
|
|
|
626
|
-
> **Not to be confused with `@tanstack/ai-code-mode-
|
|
626
|
+
> **Not to be confused with `@tanstack/ai-code-mode-snippets`**, whose snippets are TypeScript functions your application generates and runs in its own Code Mode sandbox (a local JS isolate). Provider Skills are hosted, provider-managed bundles that the model loads on demand and runs inside the provider's server-side sandbox.
|
|
627
627
|
|
|
628
628
|
Provider Skills are inert without an execution tool. The execution tool is what activates the sandbox; skills are additional capability bundles that run inside it:
|
|
629
629
|
|
|
@@ -131,7 +131,8 @@ export interface TextAdapter<
|
|
|
131
131
|
* Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,
|
|
132
132
|
* TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final
|
|
133
133
|
* `CUSTOM` event named `structured-output.complete` whose `value` is
|
|
134
|
-
* `{ object, raw, reasoning? }`.
|
|
134
|
+
* `{ object, raw, reasoning? }`. Events must be timestamped when emitted so
|
|
135
|
+
* their timestamps follow stream order.
|
|
135
136
|
*/
|
|
136
137
|
structuredOutputStream?: (
|
|
137
138
|
options: StructuredOutputOptions<TProviderOptions>,
|
|
@@ -159,6 +160,20 @@ export interface TextAdapter<
|
|
|
159
160
|
supportsCombinedToolsAndSchema?: (
|
|
160
161
|
modelOptions?: TProviderOptions | undefined,
|
|
161
162
|
) => boolean
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Where native-combined structured output is taken from.
|
|
166
|
+
*
|
|
167
|
+
* - `'text'` (default when omitted): the agent loop's accumulated
|
|
168
|
+
* assistant text is schema JSON. The engine parses it after the loop.
|
|
169
|
+
* HTTP adapters use this.
|
|
170
|
+
* - `'event'`: the adapter emits `structured-output.complete` during
|
|
171
|
+
* `chatStream`. The engine must not parse accumulated prose. Harness
|
|
172
|
+
* adapters use this.
|
|
173
|
+
*/
|
|
174
|
+
combinedStructuredOutputSource?: (
|
|
175
|
+
modelOptions?: TProviderOptions | undefined,
|
|
176
|
+
) => 'text' | 'event'
|
|
162
177
|
}
|
|
163
178
|
|
|
164
179
|
/**
|