@tanstack/ai 0.44.0 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/dist/esm/activities/chat/adapter.d.ts +13 -1
  2. package/dist/esm/activities/chat/adapter.js.map +1 -1
  3. package/dist/esm/activities/chat/index.js +126 -68
  4. package/dist/esm/activities/chat/index.js.map +1 -1
  5. package/dist/esm/activities/chat/messages.js +24 -27
  6. package/dist/esm/activities/chat/messages.js.map +1 -1
  7. package/dist/esm/activities/chat/stream/processor.js +14 -13
  8. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  9. package/dist/esm/activities/chat/tools/approval-schema.js +11 -8
  10. package/dist/esm/activities/chat/tools/approval-schema.js.map +1 -1
  11. package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
  12. package/dist/esm/activities/chat/tools/schema-converter.js.map +1 -1
  13. package/dist/esm/activities/chat/tools/tool-calls.js +40 -31
  14. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  15. package/dist/esm/activities/generateVideo/index.js +3 -2
  16. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  17. package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
  18. package/dist/esm/adapter-internals.d.ts +2 -0
  19. package/dist/esm/adapter-internals.js +3 -1
  20. package/dist/esm/interrupt-resume.js +24 -20
  21. package/dist/esm/interrupt-resume.js.map +1 -1
  22. package/dist/esm/interrupts.js +2 -1
  23. package/dist/esm/interrupts.js.map +1 -1
  24. package/dist/esm/logger/console-logger.js +1 -3
  25. package/dist/esm/logger/console-logger.js.map +1 -1
  26. package/dist/esm/logger/resolve.js +2 -1
  27. package/dist/esm/logger/resolve.js.map +1 -1
  28. package/dist/esm/stream-durability.js +1 -1
  29. package/dist/esm/stream-durability.js.map +1 -1
  30. package/dist/esm/types.d.ts +19 -16
  31. package/dist/esm/utilities/chat-params.js +1 -3
  32. package/dist/esm/utilities/chat-params.js.map +1 -1
  33. package/dist/esm/utilities/media-prompt.js +1 -3
  34. package/dist/esm/utilities/media-prompt.js.map +1 -1
  35. package/dist/esm/utilities/structured-output-events.d.ts +17 -0
  36. package/dist/esm/utilities/structured-output-events.js +32 -0
  37. package/dist/esm/utilities/structured-output-events.js.map +1 -0
  38. package/dist/esm/utilities/structured-output-text.d.ts +7 -0
  39. package/dist/esm/utilities/structured-output-text.js +50 -0
  40. package/dist/esm/utilities/structured-output-text.js.map +1 -0
  41. package/package.json +2 -2
  42. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +6 -2
  43. package/skills/ai-core/media-generation/SKILL.md +80 -34
  44. package/skills/ai-core/structured-outputs/SKILL.md +73 -1
  45. package/skills/ai-core/tool-calling/SKILL.md +1 -1
  46. package/src/activities/chat/adapter.ts +16 -1
  47. package/src/activities/chat/index.ts +155 -49
  48. package/src/activities/chat/stream/processor.ts +6 -3
  49. package/src/activities/chat/tools/tool-calls.ts +12 -4
  50. package/src/adapter-internals.ts +8 -0
  51. package/src/types.ts +27 -19
  52. package/src/utilities/structured-output-events.ts +44 -0
  53. package/src/utilities/structured-output-text.ts +63 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.44.0",
3
+ "version": "0.45.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -93,7 +93,7 @@
93
93
  },
94
94
  "devDependencies": {
95
95
  "@opentelemetry/api": "^1.9.0",
96
- "@vitest/coverage-v8": "4.0.14",
96
+ "@vitest/coverage-v8": "4.1.10",
97
97
  "arktype": "^2.1.28",
98
98
  "zod": "^4.2.0"
99
99
  },
@@ -94,8 +94,12 @@ Note: `GOOGLE_GENAI_API_KEY` does NOT work.
94
94
  ## Gotchas
95
95
 
96
96
  - All Gemini models are multimodal (text, image, audio, video, document input).
97
- - Image generation models (`gemini-3-pro-image-preview`, etc.) have smaller
98
- input limits (65K tokens) compared to text models (1M tokens).
97
+ - Image generation models (`gemini-3-pro-image`, `gemini-3.1-flash-image`, etc.)
98
+ have smaller input limits (65K tokens) compared to text models (1M tokens).
99
+ - Use the GA image ids. `gemini-3-pro-image-preview` and
100
+ `gemini-3.1-flash-image-preview` were shut down on 2026-06-25 and now 404;
101
+ they remain in the type union only as deprecated aliases, so a call to them
102
+ compiles and then fails at runtime.
99
103
  - `thinkingConfig.thinkingLevel` (level-based) and `thinkingConfig.thinkingBudget`
100
104
  (budget-based) serve different models. Check which your model supports.
101
105
  - `cachedContent` must follow the format `cachedContents/{id}`.
@@ -4,9 +4,10 @@ description: >
4
4
  Image, audio, video, speech (TTS), and transcription generation using
5
5
  activity-specific adapters: generateImage() with openaiImage/geminiImage/byteplusImage,
6
6
  generateAudio() with geminiAudio/falAudio, generateVideo() with async
7
- polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo, per-model typed
8
- durations), generateSpeech() with openaiSpeech/byteplusSpeech, generateTranscription()
9
- with openaiTranscription/byteplusTranscription. React hooks: useGenerateImage, useGenerateAudio,
7
+ polling (openaiVideo/geminiVideo/grokVideo/falVideo/byteplusVideo/openRouterVideo,
8
+ per-model typed durations), generateSpeech() with openaiSpeech/byteplusSpeech,
9
+ generateTranscription() with openaiTranscription/byteplusTranscription. React hooks:
10
+ useGenerateImage, useGenerateAudio,
10
11
  useGenerateSpeech, useTranscription, useGenerateVideo.
11
12
  TanStack Start server function integration with toServerSentEventsResponse.
12
13
  type: sub-skill
@@ -150,9 +151,16 @@ function ImageGenerator() {
150
151
  ### 1. Image Generation
151
152
 
152
153
  Supported adapters: `openaiImage` (dall-e-2, dall-e-3, gpt-image-1,
153
- gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image-preview,
154
- gemini-3.1-flash-lite-image, imagen-4.0-generate-001, etc.) and `byteplusImage`
155
- (Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`, the 5.0 family).
154
+ gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image,
155
+ gemini-3.1-flash-lite-image, gemini-3-pro-image, imagen-4.0-generate-001, etc.)
156
+ and `byteplusImage` (Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`,
157
+ the 5.0 family).
158
+
159
+ > **Use the GA Gemini image ids.** `gemini-3.1-flash-image-preview` and
160
+ > `gemini-3-pro-image-preview` were shut down on 2026-06-25 and now 404. They
161
+ > survive in the type union only as deprecated aliases so existing code keeps
162
+ > compiling — a call to them typechecks and then fails at runtime. Use
163
+ > `gemini-3.1-flash-image` / `gemini-3-pro-image` instead.
156
164
 
157
165
  > **Seedream quirks:** `watermark` defaults to **`true`** (pass
158
166
  > `modelOptions: { watermark: false }` for a clean image), `size` is a token
@@ -181,7 +189,7 @@ const openaiResult = await generateImage({
181
189
 
182
190
  // Gemini native model with aspect-ratio sizes
183
191
  const geminiResult = await generateImage({
184
- adapter: geminiImage('gemini-3.1-flash-image-preview'),
192
+ adapter: geminiImage('gemini-3.1-flash-image'),
185
193
  prompt: 'A futuristic cityscape at night',
186
194
  size: '16:9_4K',
187
195
  })
@@ -251,7 +259,8 @@ await generateImage({
251
259
  ],
252
260
  })
253
261
 
254
- // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional end_image_url)
262
+ // Image-to-video (OpenAI Sora: single input_reference; fal: image_url + optional
263
+ // end_image_url; OpenRouter: frame_images + input_references)
255
264
  import { generateVideo } from '@tanstack/ai'
256
265
  import { falVideo } from '@tanstack/ai-fal'
257
266
 
@@ -282,29 +291,30 @@ with `allowUrlFetch: true` on the adapter config
282
291
 
283
292
  **Role hints** (`metadata.role`):
284
293
 
285
- | Role | Maps to |
286
- | --------------- | ----------------------------------------------------------------------------------------------------- |
287
- | `'reference'` | fal `reference_image_urls`; Gemini multimodal part; positional otherwise |
288
- | `'character'` | Same as `'reference'`; Veo `referenceImages` slot (planned — no Veo adapter yet) |
289
- | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
290
- | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
291
- | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); Veo `image` (planned) |
292
- | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); Veo `lastFrame` (planned) |
294
+ | Role | Maps to |
295
+ | --------------- | -------------------------------------------------------------------------------------------------------------------------------------- |
296
+ | `'reference'` | fal `reference_image_urls`; OpenRouter video `input_references[]`; Gemini multimodal part; positional otherwise |
297
+ | `'character'` | Same as `'reference'`; Veo `referenceImages`; OpenRouter `input_references[]` |
298
+ | `'mask'` | OpenAI `mask` (gpt-image-2, gpt-image-1, dall-e-2); fal `mask_url` |
299
+ | `'control'` | fal `control_image_url` (ControlNet / depth / pose) |
300
+ | `'start_frame'` | fal `start_image_url` (or the endpoint's field, e.g. `image_url` on Kling i2v); OpenRouter `frame_images[]` `first_frame`; Veo `image` |
301
+ | `'end_frame'` | fal `end_image_url` (or e.g. `tail_image_url` / `last_frame_url`); OpenRouter `frame_images[]` `last_frame`; Veo `lastFrame` |
293
302
 
294
303
  **Provider support matrix:**
295
304
 
296
- | Provider | `generateImage` image parts | `generateVideo` image parts |
297
- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
298
- | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
299
- | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | No native Veo adapter yet — deferred to a follow-up. |
300
- | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
301
- | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
302
- | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | n/a |
303
- | Anthropic | n/a (no image generation API). | n/a |
305
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
306
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
307
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
308
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
309
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
310
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). Starting frame and reference inputs cannot be combined. A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
311
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
312
+ | Anthropic | n/a (no image generation API). | n/a |
304
313
 
305
314
  Video and audio prompt parts follow the same `metadata.role` convention
306
- for video-to-video and lipsync flows on fal; other providers throw when
307
- they're passed.
315
+ for video-to-video and lipsync flows on fal. Grok accepts one source
316
+ `video` part on `grok-imagine-video` with `modelOptions.mode: 'edit' | 'extend'`
317
+ and rejects audio parts. Other providers throw when those parts are passed.
308
318
 
309
319
  ### 2. Audio Generation (Music, Sound Effects)
310
320
 
@@ -445,7 +455,13 @@ const { generate, result, isLoading } = useTranscription({
445
455
  ### 5. Video Generation (Experimental -- async polling)
446
456
 
447
457
  Video generation uses a jobs/polling architecture. The server creates a job,
448
- polls for status, and streams updates to the client.
458
+ polls for status, and streams updates to the client. Adapters: `openaiVideo`
459
+ (Sora), `geminiVideo` (Veo / Omni Flash), `grokVideo`, `byteplusVideo`
460
+ (Seedance), `falVideo` (Kling, MiniMax, Hunyuan, …), and `openRouterVideo`
461
+ (OpenRouter's dedicated `POST /api/v1/videos` gateway — Seedance, Veo, Wan,
462
+ Kling, Sora 2 Pro and others through one API key; `getVideoJobStatus()`
463
+ returns the video as a `data:` URL since OpenRouter's download URLs require
464
+ the API key, and surfaces the gateway-reported cost as `usage.cost`).
449
465
 
450
466
  ```typescript
451
467
  import {
@@ -536,13 +552,20 @@ const edited = await generateVideo({
536
552
 
537
553
  Other video adapters: `openaiVideo('sora-2')` (pixel sizes like `'1280x720'`,
538
554
  durations 4/8/12s, single `input_reference` image prompt part), `grokVideo(...)`
539
- (`grok-imagine-video` does text-to-video + image-to-video; `grok-imagine-video-1.5` is
540
- image-to-video only — needs an `image` prompt part as the starting frame, text-only throws;
541
- aspect-ratio size template like `'16:9_720p'`, integer durations 1-15s, reports
542
- `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
555
+ (`grok-imagine-video` and `grok-imagine-video-1.5` both do text-to-video + image-to-video;
556
+ 1.5 adds reference-to-video — `'reference'`/`'character'`-roled image parts →
557
+ `reference_images` (max 7), preset voices via `modelOptions.reference_audios` (max 3) —
558
+ 1.5-only, capped at 720p, and not combinable with a starting-frame image; only
559
+ `grok-imagine-video` edits/extends a source `video` prompt part via
560
+ `modelOptions.mode: 'edit' | 'extend'` (extend `duration` = added tail). Edit/extend
561
+ outputs inherit the source clip's properties, so `size`/`aspect_ratio`/`resolution`
562
+ throw in both modes and `duration` throws in edit mode — pass none of them there;
563
+ generation uses the aspect-ratio size template like `'16:9_720p'` (1080p is 1.5-only),
564
+ integer durations 1-15s, reports `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
543
565
  aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family,
544
- 4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`), and
545
- `falVideo(...)` (hosted models, see cost tracking below).
566
+ 4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
567
+ `openRouterVideo(...)` (OpenRouter's dedicated `POST /api/v1/videos` gateway),
568
+ and `falVideo(...)` (hosted models, see cost tracking below).
546
569
 
547
570
  > **Seedance option applicability is per model and enforced server-side** —
548
571
  > Ark returns a 400 for an inapplicable field rather than ignoring it.
@@ -554,6 +577,29 @@ aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family
554
577
  > days). Seedance is also reachable via `falVideo` — `byteplusVideo` is the
555
578
  > direct-to-BytePlus path.
556
579
 
580
+ OpenRouter (`@tanstack/ai-openrouter`, `openRouterVideo`) runs the dedicated
581
+ async video API (`POST /api/v1/videos`) and shares the same typed-duration
582
+ contract — `duration`, `size`, and provider options are narrowed per model
583
+ from OpenRouter's published metadata, with the same `availableDurations()` /
584
+ `snapDuration()` helpers:
585
+
586
+ ```typescript
587
+ import { openRouterVideo } from '@tanstack/ai-openrouter'
588
+
589
+ const adapter = openRouterVideo('bytedance/seedance-2.0')
590
+ adapter.availableDurations()
591
+ // { kind: 'discrete', values: [4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15] }
592
+ adapter.snapDuration(7.4) // 7
593
+
594
+ const sliderSeconds = 7 // raw seconds from a UI control
595
+ const { jobId } = await generateVideo({
596
+ adapter,
597
+ prompt: 'A timelapse of clouds',
598
+ duration: adapter.snapDuration(sliderSeconds),
599
+ })
600
+ // Completed url is a data: URL; usage.cost carries the real billed cost.
601
+ ```
602
+
557
603
  Client hook with job tracking:
558
604
 
559
605
  ```tsx
@@ -964,7 +1010,7 @@ generateImage({
964
1010
  })
965
1011
 
966
1012
  generateImage({
967
- adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
1013
+ adapter: geminiImage('gemini-3.1-flash-image'), // native multimodal
968
1014
  prompt: [
969
1015
  { type: 'text', content: 'Edit this' },
970
1016
  { type: 'image', source: { type: 'url', value: url } },
@@ -20,6 +20,7 @@ sources:
20
20
  - 'TanStack/ai:docs/structured-outputs/streaming.md'
21
21
  - 'TanStack/ai:docs/structured-outputs/multi-turn.md'
22
22
  - 'TanStack/ai:docs/structured-outputs/with-tools.md'
23
+ - 'TanStack/ai:docs/structured-outputs/harnesses.md'
23
24
  ---
24
25
 
25
26
  # Structured Outputs
@@ -48,7 +49,7 @@ person.age // number
48
49
 
49
50
  When `outputSchema` is provided, `chat()` returns `Promise<InferSchemaType<TSchema>>` instead of `AsyncIterable<StreamChunk>`. The result is fully typed.
50
51
 
51
- Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client, and **Pattern 5** for multi-turn structured chats.
52
+ Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client, **Pattern 5** for multi-turn structured chats, and **Pattern 6** for harness adapters.
52
53
 
53
54
  ## Decision: which pattern fits
54
55
 
@@ -59,6 +60,7 @@ Adding `stream: true` switches the return to `StructuredOutputStream<InferSchema
59
60
  | Direct iteration of the stream in Node or tests | Pattern 3 — async iterable |
60
61
  | Users iterate on a structured object across multiple turns (recipe builder, ticket refinement) | Pattern 5 — multi-turn structured chat |
61
62
  | Tools that gather info, then return a typed object | Combine any of the above with `tools` — see ai-core/tool-calling |
63
+ | A coding agent in a sandbox inspects files, then returns a typed object | Pattern 6 — harness `outputSchema` |
62
64
 
63
65
  ## Core Patterns
64
66
 
@@ -189,6 +191,13 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
189
191
  | `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
190
192
  | `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
191
193
  | `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
194
+ | `@tanstack/ai-bedrock` | Separate native `structuredOutputStream` finalization through Converse or an OpenAI-compatible API |
195
+ | `@tanstack/ai-byteplus` | Native combined mode on supported models; unsupported models emit `RUN_ERROR` |
196
+ | `@tanstack/ai-claude-code` | Combined + event source — `--json-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
197
+ | `@tanstack/ai-codex` | Combined + event source — `--output-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
198
+ | `@tanstack/ai-opencode` | Combined + event source — prompt-and-parse. Read `useChat().final`. See Pattern 6. |
199
+ | `@tanstack/ai-grok-build` | Combined + event source — prompt-and-parse (ACP and streaming-json). Read `useChat().final` or the `structured-output` part. See Pattern 6. |
200
+ | `@tanstack/ai-acp` (`acpCompatible`) | Combined + event source — prompt-and-parse. Read `useChat().final` or the `structured-output` part. See Pattern 6. |
192
201
  | All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
193
202
 
194
203
  **Native combined mode vs fallback** is signaled by the adapter's
@@ -341,6 +350,68 @@ Key behaviors:
341
350
  - **`partial` / `final` are derived.** The hook-level `partial` and `final` are NOT singleton state — they're derived from the latest assistant message's part (the one after the most recent user message). Between `sendMessage()` and the first chunk, `partial` reads `{}` and `final` reads `null` because no new assistant turn exists yet.
342
351
  - **Round-trip preserves history.** When the client sends turn N+1, each prior assistant turn's `structured-output` part is serialized back as `{ role: 'assistant', content: <part.raw> }` so the model sees its own prior structured response. Streaming / errored parts are dropped from the round-trip.
343
352
 
353
+ ### Pattern 6: Harness adapters (Claude Code, Codex, OpenCode, Grok Build, ACP)
354
+
355
+ Dedicated harness adapters honor `chat({ outputSchema })` on the same turn. Native harness tools still run. Read the object from `await chat()`, from `useChat().final`, or from the assistant `structured-output` part on `messages[].parts`. Do not parse assistant prose.
356
+
357
+ A UI endpoint must pass `stream: true`. Without it, `chat()` returns a `Promise`, not SSE.
358
+
359
+ ```typescript
360
+ import { chat, toServerSentEventsResponse } from '@tanstack/ai'
361
+ import { claudeCodeText } from '@tanstack/ai-claude-code'
362
+ import { withSandbox } from '@tanstack/ai-sandbox'
363
+ import { z } from 'zod'
364
+ import { sandbox } from './sandbox'
365
+
366
+ const ReportSchema = z.object({
367
+ name: z.string(),
368
+ oneLiner: z.string(),
369
+ })
370
+
371
+ export async function POST(request: Request) {
372
+ const body: unknown = await request.json()
373
+ const messages =
374
+ typeof body === 'object' &&
375
+ body !== null &&
376
+ 'messages' in body &&
377
+ Array.isArray(body.messages)
378
+ ? body.messages
379
+ : []
380
+
381
+ const stream = chat({
382
+ adapter: claudeCodeText('claude-opus-4-8'),
383
+ messages,
384
+ outputSchema: ReportSchema,
385
+ stream: true,
386
+ middleware: [withSandbox(sandbox)],
387
+ })
388
+ return toServerSentEventsResponse(stream)
389
+ }
390
+ ```
391
+
392
+ ```tsx
393
+ import { useChat, fetchServerSentEvents } from '@tanstack/ai-react'
394
+ import { z } from 'zod'
395
+
396
+ const ReportSchema = z.object({
397
+ name: z.string(),
398
+ oneLiner: z.string(),
399
+ })
400
+
401
+ const { final } = useChat({
402
+ connection: fetchServerSentEvents('/api/repo-report'),
403
+ outputSchema: ReportSchema,
404
+ })
405
+
406
+ final?.name
407
+ ```
408
+
409
+ - Claude Code: `--json-schema`. Codex: `--output-schema`. OpenCode, Grok Build, and `acpCompatible`: prompt-and-parse.
410
+ - `partial` stays empty until `structured-output.complete`.
411
+ - Client tools and `needsApproval` fail fast. The harness cannot pause for a browser round-trip.
412
+ - Render live work from `messages[].parts` (`thinking`, `tool-call`, `text`, `structured-output`). `final` is only the latest turn.
413
+ - See [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md).
414
+
344
415
  ## Common Mistakes
345
416
 
346
417
  ### HIGH: Filtering `TextPart`s out of `useChat` renderers when using `outputSchema`
@@ -509,4 +580,5 @@ provider call, stripping system prompts), use the dedicated
509
580
  - See also: **ai-core/chat-experience/SKILL.md** — Base `useChat` surface; the structured-output additions documented here layer on top.
510
581
  - See also: **ai-core/adapter-configuration/SKILL.md** — Adapter handles structured-output strategy transparently.
511
582
  - See also: **ai-core/tool-calling/SKILL.md** — Combine `tools` with `outputSchema` for an agent loop that runs tools first and returns a typed object. Tool-approval and client-tool flows compose with structured runs without extra wiring; see [docs/structured-outputs/with-tools.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/with-tools.md).
583
+ - See also: [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md) — dedicated harness adapters and `useChat().final`.
512
584
  - See also: **ai-core/middleware/SKILL.md** — `onStructuredOutputConfig` hook and the `structuredOutput` phase for observing/transforming the final structured-output call.
@@ -623,7 +623,7 @@ export const Route = createFileRoute('/api/chat')({
623
623
 
624
624
  ## Provider Skills
625
625
 
626
- > **Not to be confused with `@tanstack/ai-code-mode-skills`**, which are locally-generated TypeScript functions executed client-side. Provider Skills are hosted, provider-managed bundles that the model loads on demand and runs inside the provider's server-side sandbox.
626
+ > **Not to be confused with `@tanstack/ai-code-mode-snippets`**, whose snippets are TypeScript functions your application generates and runs in its own Code Mode sandbox (a local JS isolate). Provider Skills are hosted, provider-managed bundles that the model loads on demand and runs inside the provider's server-side sandbox.
627
627
 
628
628
  Provider Skills are inert without an execution tool. The execution tool is what activates the sandbox; skills are additional capability bundles that run inside it:
629
629
 
@@ -131,7 +131,8 @@ export interface TextAdapter<
131
131
  * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,
132
132
  * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final
133
133
  * `CUSTOM` event named `structured-output.complete` whose `value` is
134
- * `{ object, raw, reasoning? }`.
134
+ * `{ object, raw, reasoning? }`. Events must be timestamped when emitted so
135
+ * their timestamps follow stream order.
135
136
  */
136
137
  structuredOutputStream?: (
137
138
  options: StructuredOutputOptions<TProviderOptions>,
@@ -159,6 +160,20 @@ export interface TextAdapter<
159
160
  supportsCombinedToolsAndSchema?: (
160
161
  modelOptions?: TProviderOptions | undefined,
161
162
  ) => boolean
163
+
164
+ /**
165
+ * Where native-combined structured output is taken from.
166
+ *
167
+ * - `'text'` (default when omitted): the agent loop's accumulated
168
+ * assistant text is schema JSON. The engine parses it after the loop.
169
+ * HTTP adapters use this.
170
+ * - `'event'`: the adapter emits `structured-output.complete` during
171
+ * `chatStream`. The engine must not parse accumulated prose. Harness
172
+ * adapters use this.
173
+ */
174
+ combinedStructuredOutputSource?: (
175
+ modelOptions?: TProviderOptions | undefined,
176
+ ) => 'text' | 'event'
162
177
  }
163
178
 
164
179
  /**