@tanstack/ai 0.44.1 → 0.45.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/dist/esm/activities/chat/adapter.d.ts +13 -1
  2. package/dist/esm/activities/chat/adapter.js.map +1 -1
  3. package/dist/esm/activities/chat/index.js +108 -68
  4. package/dist/esm/activities/chat/index.js.map +1 -1
  5. package/dist/esm/activities/chat/messages.js +24 -27
  6. package/dist/esm/activities/chat/messages.js.map +1 -1
  7. package/dist/esm/activities/chat/stream/processor.js +14 -13
  8. package/dist/esm/activities/chat/stream/processor.js.map +1 -1
  9. package/dist/esm/activities/chat/tools/approval-schema.js +11 -8
  10. package/dist/esm/activities/chat/tools/approval-schema.js.map +1 -1
  11. package/dist/esm/activities/chat/tools/lazy-tool-manager.js.map +1 -1
  12. package/dist/esm/activities/chat/tools/schema-converter.js.map +1 -1
  13. package/dist/esm/activities/chat/tools/tool-calls.js +40 -31
  14. package/dist/esm/activities/chat/tools/tool-calls.js.map +1 -1
  15. package/dist/esm/activities/generateVideo/index.js +3 -2
  16. package/dist/esm/activities/generateVideo/index.js.map +1 -1
  17. package/dist/esm/activities/summarize/chat-stream-summarize.js.map +1 -1
  18. package/dist/esm/adapter-internals.d.ts +2 -0
  19. package/dist/esm/adapter-internals.js +3 -1
  20. package/dist/esm/interrupt-resume.js +24 -20
  21. package/dist/esm/interrupt-resume.js.map +1 -1
  22. package/dist/esm/interrupts.js +2 -1
  23. package/dist/esm/interrupts.js.map +1 -1
  24. package/dist/esm/logger/console-logger.js +1 -3
  25. package/dist/esm/logger/console-logger.js.map +1 -1
  26. package/dist/esm/logger/resolve.js +2 -1
  27. package/dist/esm/logger/resolve.js.map +1 -1
  28. package/dist/esm/stream-durability.js +1 -1
  29. package/dist/esm/stream-durability.js.map +1 -1
  30. package/dist/esm/types.d.ts +19 -16
  31. package/dist/esm/utilities/chat-params.js +1 -3
  32. package/dist/esm/utilities/chat-params.js.map +1 -1
  33. package/dist/esm/utilities/media-prompt.js +1 -3
  34. package/dist/esm/utilities/media-prompt.js.map +1 -1
  35. package/dist/esm/utilities/structured-output-events.d.ts +17 -0
  36. package/dist/esm/utilities/structured-output-events.js +32 -0
  37. package/dist/esm/utilities/structured-output-events.js.map +1 -0
  38. package/dist/esm/utilities/structured-output-text.d.ts +7 -0
  39. package/dist/esm/utilities/structured-output-text.js +50 -0
  40. package/dist/esm/utilities/structured-output-text.js.map +1 -0
  41. package/package.json +2 -2
  42. package/skills/ai-core/adapter-configuration/references/gemini-adapter.md +6 -2
  43. package/skills/ai-core/media-generation/SKILL.md +33 -19
  44. package/skills/ai-core/structured-outputs/SKILL.md +73 -1
  45. package/skills/ai-core/tool-calling/SKILL.md +1 -1
  46. package/src/activities/chat/adapter.ts +16 -1
  47. package/src/activities/chat/index.ts +128 -47
  48. package/src/activities/chat/stream/processor.ts +6 -3
  49. package/src/activities/chat/tools/tool-calls.ts +12 -4
  50. package/src/adapter-internals.ts +8 -0
  51. package/src/types.ts +27 -19
  52. package/src/utilities/structured-output-events.ts +44 -0
  53. package/src/utilities/structured-output-text.ts +63 -0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tanstack/ai",
3
- "version": "0.44.1",
3
+ "version": "0.45.0",
4
4
  "description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
5
5
  "author": "Tanner Linsley",
6
6
  "license": "MIT",
@@ -93,7 +93,7 @@
93
93
  },
94
94
  "devDependencies": {
95
95
  "@opentelemetry/api": "^1.9.0",
96
- "@vitest/coverage-v8": "4.0.14",
96
+ "@vitest/coverage-v8": "4.1.10",
97
97
  "arktype": "^2.1.28",
98
98
  "zod": "^4.2.0"
99
99
  },
@@ -94,8 +94,12 @@ Note: `GOOGLE_GENAI_API_KEY` does NOT work.
94
94
  ## Gotchas
95
95
 
96
96
  - All Gemini models are multimodal (text, image, audio, video, document input).
97
- - Image generation models (`gemini-3-pro-image-preview`, etc.) have smaller
98
- input limits (65K tokens) compared to text models (1M tokens).
97
+ - Image generation models (`gemini-3-pro-image`, `gemini-3.1-flash-image`, etc.)
98
+ have smaller input limits (65K tokens) compared to text models (1M tokens).
99
+ - Use the GA image ids. `gemini-3-pro-image-preview` and
100
+ `gemini-3.1-flash-image-preview` were shut down on 2026-06-25 and now 404;
101
+ they remain in the type union only as deprecated aliases, so a call to them
102
+ compiles and then fails at runtime.
99
103
  - `thinkingConfig.thinkingLevel` (level-based) and `thinkingConfig.thinkingBudget`
100
104
  (budget-based) serve different models. Check which your model supports.
101
105
  - `cachedContent` must follow the format `cachedContents/{id}`.
@@ -151,9 +151,16 @@ function ImageGenerator() {
151
151
  ### 1. Image Generation
152
152
 
153
153
  Supported adapters: `openaiImage` (dall-e-2, dall-e-3, gpt-image-1,
154
- gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image-preview,
155
- gemini-3.1-flash-lite-image, imagen-4.0-generate-001, etc.) and `byteplusImage`
156
- (Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`, the 5.0 family).
154
+ gpt-image-1-mini, gpt-image-2), `geminiImage` (gemini-3.1-flash-image,
155
+ gemini-3.1-flash-lite-image, gemini-3-pro-image, imagen-4.0-generate-001, etc.)
156
+ and `byteplusImage` (Seedream — `seedream-4-0-250828`, `seedream-4-5-251128`,
157
+ the 5.0 family).
158
+
159
+ > **Use the GA Gemini image ids.** `gemini-3.1-flash-image-preview` and
160
+ > `gemini-3-pro-image-preview` were shut down on 2026-06-25 and now 404. They
161
+ > survive in the type union only as deprecated aliases so existing code keeps
162
+ > compiling — a call to them typechecks and then fails at runtime. Use
163
+ > `gemini-3.1-flash-image` / `gemini-3-pro-image` instead.
157
164
 
158
165
  > **Seedream quirks:** `watermark` defaults to **`true`** (pass
159
166
  > `modelOptions: { watermark: false }` for a clean image), `size` is a token
@@ -182,7 +189,7 @@ const openaiResult = await generateImage({
182
189
 
183
190
  // Gemini native model with aspect-ratio sizes
184
191
  const geminiResult = await generateImage({
185
- adapter: geminiImage('gemini-3.1-flash-image-preview'),
192
+ adapter: geminiImage('gemini-3.1-flash-image'),
186
193
  prompt: 'A futuristic cityscape at night',
187
194
  size: '16:9_4K',
188
195
  })
@@ -295,18 +302,19 @@ with `allowUrlFetch: true` on the adapter config
295
302
 
296
303
  **Provider support matrix:**
297
304
 
298
- | Provider | `generateImage` image parts | `generateVideo` image parts |
299
- | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
300
- | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
301
- | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
302
- | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
303
- | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | n/a |
304
- | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
305
- | Anthropic | n/a (no image generation API). | n/a |
305
+ | Provider | `generateImage` image parts | `generateVideo` image parts |
306
+ | ---------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
307
+ | OpenAI | gpt-image-2 / gpt-image-1 / -mini → `images.edit()` (up to 16). dall-e-2 → edit (1). dall-e-3 throws. | Sora-2 / -pro → `input_reference` (single). Throws if >1. |
308
+ | Gemini | Native (gemini-\*-flash-image, "nano-banana") → multimodal `contents`. Imagen throws. | Veo → first un-roled / `'start_frame'` image is the input image; `'end_frame'` → `lastFrame`; `'reference'` / `'character'` → `referenceImages`. Omni Flash sends image/video parts as interaction content blocks (no role routing). |
309
+ | fal | Per-endpoint field names from a generated map (`pnpm generate:fal-image-fields`). Defaults: 1 input → `image_url`; >1 → `image_urls`; roles → `mask_url` / `control_image_url` / `reference_image_urls`. | Per-endpoint map (e.g. Kling i2v start frame → `image_url`). Defaults: 1 input → `image_url`; `start_frame`/`end_frame` → `start_image_url`/`end_image_url`; `reference` → `reference_image_urls`. |
310
+ | Grok | grok-imagine models → `/v1/images/edits` JSON endpoint (≤3 sources, addressed by xAI in request order; prompt sent verbatim; mask/control throw). grok-2-image-1212 throws. | Un-roled / `'start_frame'` image → starting frame; `'reference'` / `'character'` → `reference_images` (1.5). Starting frame and reference inputs cannot be combined. A `video` part + `modelOptions.mode: 'edit' \| 'extend'` routes to `/videos/edits` / `/videos/extensions` on `grok-imagine-video` only. |
311
+ | OpenRouter | Prompt parts map 1:1 onto multimodal `text` / `image_url` content parts, preserving interleaved order. | Dedicated async API (`openRouterVideo`): `start_frame`/`end_frame` → `frame_images[]` (`first_frame`/`last_frame`); `reference`/`character` → `input_references[]`; an unroled image defaults to the start frame. Frame roles validated against the model's `supported_frame_images` metadata. |
312
+ | Anthropic | n/a (no image generation API). | n/a |
306
313
 
307
314
  Video and audio prompt parts follow the same `metadata.role` convention
308
- for video-to-video and lipsync flows on fal; other providers throw when
309
- they're passed.
315
+ for video-to-video and lipsync flows on fal. Grok accepts one source
316
+ `video` part on `grok-imagine-video` with `modelOptions.mode: 'edit' | 'extend'`
317
+ and rejects audio parts. Other providers throw when those parts are passed.
310
318
 
311
319
  ### 2. Audio Generation (Music, Sound Effects)
312
320
 
@@ -544,10 +552,16 @@ const edited = await generateVideo({
544
552
 
545
553
  Other video adapters: `openaiVideo('sora-2')` (pixel sizes like `'1280x720'`,
546
554
  durations 4/8/12s, single `input_reference` image prompt part), `grokVideo(...)`
547
- (`grok-imagine-video` does text-to-video + image-to-video; `grok-imagine-video-1.5` is
548
- image-to-video only — needs an `image` prompt part as the starting frame, text-only throws;
549
- aspect-ratio size template like `'16:9_720p'`, integer durations 1-15s, reports
550
- `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
555
+ (`grok-imagine-video` and `grok-imagine-video-1.5` both do text-to-video + image-to-video;
556
+ 1.5 adds reference-to-video — `'reference'`/`'character'`-roled image parts →
557
+ `reference_images` (max 7), preset voices via `modelOptions.reference_audios` (max 3) —
558
+ 1.5-only, capped at 720p, and not combinable with a starting-frame image; only
559
+ `grok-imagine-video` edits/extends a source `video` prompt part via
560
+ `modelOptions.mode: 'edit' | 'extend'` (extend `duration` = added tail). Edit/extend
561
+ outputs inherit the source clip's properties, so `size`/`aspect_ratio`/`resolution`
562
+ throw in both modes and `duration` throws in edit mode — pass none of them there;
563
+ generation uses the aspect-ratio size template like `'16:9_720p'` (1080p is 1.5-only),
564
+ integer durations 1-15s, reports `usage.unitsBilled` seconds and exact `usage.cost`), `byteplusVideo(...)` (Seedance —
551
565
  aspect-ratio size template like `'16:9_720p'`, durations 4-15s on the 2.0 family,
552
566
  4-12s on 1.5-pro, 2-12s on the 1.0-pro models; reads `ARK_API_KEY`),
553
567
  `openRouterVideo(...)` (OpenRouter's dedicated `POST /api/v1/videos` gateway),
@@ -996,7 +1010,7 @@ generateImage({
996
1010
  })
997
1011
 
998
1012
  generateImage({
999
- adapter: geminiImage('gemini-3.1-flash-image-preview'), // native multimodal
1013
+ adapter: geminiImage('gemini-3.1-flash-image'), // native multimodal
1000
1014
  prompt: [
1001
1015
  { type: 'text', content: 'Edit this' },
1002
1016
  { type: 'image', source: { type: 'url', value: url } },
@@ -20,6 +20,7 @@ sources:
20
20
  - 'TanStack/ai:docs/structured-outputs/streaming.md'
21
21
  - 'TanStack/ai:docs/structured-outputs/multi-turn.md'
22
22
  - 'TanStack/ai:docs/structured-outputs/with-tools.md'
23
+ - 'TanStack/ai:docs/structured-outputs/harnesses.md'
23
24
  ---
24
25
 
25
26
  # Structured Outputs
@@ -48,7 +49,7 @@ person.age // number
48
49
 
49
50
  When `outputSchema` is provided, `chat()` returns `Promise<InferSchemaType<TSchema>>` instead of `AsyncIterable<StreamChunk>`. The result is fully typed.
50
51
 
51
- Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client, and **Pattern 5** for multi-turn structured chats.
52
+ Adding `stream: true` switches the return to `StructuredOutputStream<InferSchemaType<TSchema>>` — incremental JSON deltas plus a terminal validated object. See **Pattern 3** below for direct iteration, **Pattern 4** for the `useChat` shape on the client, **Pattern 5** for multi-turn structured chats, and **Pattern 6** for harness adapters.
52
53
 
53
54
  ## Decision: which pattern fits
54
55
 
@@ -59,6 +60,7 @@ Adding `stream: true` switches the return to `StructuredOutputStream<InferSchema
59
60
  | Direct iteration of the stream in Node or tests | Pattern 3 — async iterable |
60
61
  | Users iterate on a structured object across multiple turns (recipe builder, ticket refinement) | Pattern 5 — multi-turn structured chat |
61
62
  | Tools that gather info, then return a typed object | Combine any of the above with `tools` — see ai-core/tool-calling |
63
+ | A coding agent in a sandbox inspects files, then returns a typed object | Pattern 6 — harness `outputSchema` |
62
64
 
63
65
  ## Core Patterns
64
66
 
@@ -189,6 +191,13 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
189
191
  | `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
190
192
  | `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
191
193
  | `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
194
+ | `@tanstack/ai-bedrock` | Separate native `structuredOutputStream` finalization through Converse or an OpenAI-compatible API |
195
+ | `@tanstack/ai-byteplus` | Native combined mode on supported models; unsupported models emit `RUN_ERROR` |
196
+ | `@tanstack/ai-claude-code` | Combined + event source — `--json-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
197
+ | `@tanstack/ai-codex` | Combined + event source — `--output-schema` on the same harness turn. Read `useChat().final`. See Pattern 6. |
198
+ | `@tanstack/ai-opencode` | Combined + event source — prompt-and-parse. Read `useChat().final`. See Pattern 6. |
199
+ | `@tanstack/ai-grok-build` | Combined + event source — prompt-and-parse (ACP and streaming-json). Read `useChat().final` or the `structured-output` part. See Pattern 6. |
200
+ | `@tanstack/ai-acp` (`acpCompatible`) | Combined + event source — prompt-and-parse. Read `useChat().final` or the `structured-output` part. See Pattern 6. |
192
201
  | All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
193
202
 
194
203
  **Native combined mode vs fallback** is signaled by the adapter's
@@ -341,6 +350,68 @@ Key behaviors:
341
350
  - **`partial` / `final` are derived.** The hook-level `partial` and `final` are NOT singleton state — they're derived from the latest assistant message's part (the one after the most recent user message). Between `sendMessage()` and the first chunk, `partial` reads `{}` and `final` reads `null` because no new assistant turn exists yet.
342
351
  - **Round-trip preserves history.** When the client sends turn N+1, each prior assistant turn's `structured-output` part is serialized back as `{ role: 'assistant', content: <part.raw> }` so the model sees its own prior structured response. Streaming / errored parts are dropped from the round-trip.
343
352
 
353
+ ### Pattern 6: Harness adapters (Claude Code, Codex, OpenCode, Grok Build, ACP)
354
+
355
+ Dedicated harness adapters honor `chat({ outputSchema })` on the same turn. Native harness tools still run. Read the object from `await chat()`, from `useChat().final`, or from the assistant `structured-output` part on `messages[].parts`. Do not parse assistant prose.
356
+
357
+ A UI endpoint must pass `stream: true`. Without it, `chat()` returns a `Promise`, not SSE.
358
+
359
+ ```typescript
360
+ import { chat, toServerSentEventsResponse } from '@tanstack/ai'
361
+ import { claudeCodeText } from '@tanstack/ai-claude-code'
362
+ import { withSandbox } from '@tanstack/ai-sandbox'
363
+ import { z } from 'zod'
364
+ import { sandbox } from './sandbox'
365
+
366
+ const ReportSchema = z.object({
367
+ name: z.string(),
368
+ oneLiner: z.string(),
369
+ })
370
+
371
+ export async function POST(request: Request) {
372
+ const body: unknown = await request.json()
373
+ const messages =
374
+ typeof body === 'object' &&
375
+ body !== null &&
376
+ 'messages' in body &&
377
+ Array.isArray(body.messages)
378
+ ? body.messages
379
+ : []
380
+
381
+ const stream = chat({
382
+ adapter: claudeCodeText('claude-opus-4-8'),
383
+ messages,
384
+ outputSchema: ReportSchema,
385
+ stream: true,
386
+ middleware: [withSandbox(sandbox)],
387
+ })
388
+ return toServerSentEventsResponse(stream)
389
+ }
390
+ ```
391
+
392
+ ```tsx
393
+ import { useChat, fetchServerSentEvents } from '@tanstack/ai-react'
394
+ import { z } from 'zod'
395
+
396
+ const ReportSchema = z.object({
397
+ name: z.string(),
398
+ oneLiner: z.string(),
399
+ })
400
+
401
+ const { final } = useChat({
402
+ connection: fetchServerSentEvents('/api/repo-report'),
403
+ outputSchema: ReportSchema,
404
+ })
405
+
406
+ final?.name
407
+ ```
408
+
409
+ - Claude Code: `--json-schema`. Codex: `--output-schema`. OpenCode, Grok Build, and `acpCompatible`: prompt-and-parse.
410
+ - `partial` stays empty until `structured-output.complete`.
411
+ - Client tools and `needsApproval` fail fast. The harness cannot pause for a browser round-trip.
412
+ - Render live work from `messages[].parts` (`thinking`, `tool-call`, `text`, `structured-output`). `final` is only the latest turn.
413
+ - See [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md).
414
+
344
415
  ## Common Mistakes
345
416
 
346
417
  ### HIGH: Filtering `TextPart`s out of `useChat` renderers when using `outputSchema`
@@ -509,4 +580,5 @@ provider call, stripping system prompts), use the dedicated
509
580
  - See also: **ai-core/chat-experience/SKILL.md** — Base `useChat` surface; the structured-output additions documented here layer on top.
510
581
  - See also: **ai-core/adapter-configuration/SKILL.md** — Adapter handles structured-output strategy transparently.
511
582
  - See also: **ai-core/tool-calling/SKILL.md** — Combine `tools` with `outputSchema` for an agent loop that runs tools first and returns a typed object. Tool-approval and client-tool flows compose with structured runs without extra wiring; see [docs/structured-outputs/with-tools.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/with-tools.md).
583
+ - See also: [docs/structured-outputs/harnesses.md](https://github.com/TanStack/ai/blob/main/docs/structured-outputs/harnesses.md) — dedicated harness adapters and `useChat().final`.
512
584
  - See also: **ai-core/middleware/SKILL.md** — `onStructuredOutputConfig` hook and the `structuredOutput` phase for observing/transforming the final structured-output call.
@@ -623,7 +623,7 @@ export const Route = createFileRoute('/api/chat')({
623
623
 
624
624
  ## Provider Skills
625
625
 
626
- > **Not to be confused with `@tanstack/ai-code-mode-skills`**, which are locally-generated TypeScript functions executed client-side. Provider Skills are hosted, provider-managed bundles that the model loads on demand and runs inside the provider's server-side sandbox.
626
+ > **Not to be confused with `@tanstack/ai-code-mode-snippets`**, whose snippets are TypeScript functions your application generates and runs in its own Code Mode sandbox (a local JS isolate). Provider Skills are hosted, provider-managed bundles that the model loads on demand and runs inside the provider's server-side sandbox.
627
627
 
628
628
  Provider Skills are inert without an execution tool. The execution tool is what activates the sandbox; skills are additional capability bundles that run inside it:
629
629
 
@@ -131,7 +131,8 @@ export interface TextAdapter<
131
131
  * Implementations must emit standard AG-UI lifecycle events (RUN_STARTED,
132
132
  * TEXT_MESSAGE_*, RUN_FINISHED) carrying raw JSON text deltas, plus a final
133
133
  * `CUSTOM` event named `structured-output.complete` whose `value` is
134
- * `{ object, raw, reasoning? }`.
134
+ * `{ object, raw, reasoning? }`. Events must be timestamped when emitted so
135
+ * their timestamps follow stream order.
135
136
  */
136
137
  structuredOutputStream?: (
137
138
  options: StructuredOutputOptions<TProviderOptions>,
@@ -159,6 +160,20 @@ export interface TextAdapter<
159
160
  supportsCombinedToolsAndSchema?: (
160
161
  modelOptions?: TProviderOptions | undefined,
161
162
  ) => boolean
163
+
164
+ /**
165
+ * Where native-combined structured output is taken from.
166
+ *
167
+ * - `'text'` (default when omitted): the agent loop's accumulated
168
+ * assistant text is schema JSON. The engine parses it after the loop.
169
+ * HTTP adapters use this.
170
+ * - `'event'`: the adapter emits `structured-output.complete` during
171
+ * `chatStream`. The engine must not parse accumulated prose. Harness
172
+ * adapters use this.
173
+ */
174
+ combinedStructuredOutputSource?: (
175
+ modelOptions?: TProviderOptions | undefined,
176
+ ) => 'text' | 'event'
162
177
  }
163
178
 
164
179
  /**
@@ -655,11 +655,12 @@ interface TextEngineConfig<
655
655
  * - nativeCombined: when true, the adapter declared
656
656
  * `supportsCombinedToolsAndSchema()` and the engine wires `jsonSchema`
657
657
  * into the regular `chatStream` call instead of running a separate
658
- * finalization round-trip. The agent loop's final-turn text is the
659
- * schema-constrained JSON; the engine parses it from accumulated
660
- * content. The `'structuredOutput'` middleware phase does NOT fire on
661
- * this path — middleware sees the run through `beforeModel` /
662
- * `modelStream` as usual.
658
+ * finalization round-trip. The `'structuredOutput'` middleware phase
659
+ * does NOT fire on this path — middleware sees the run through
660
+ * `beforeModel` / `modelStream` as usual.
661
+ * - source: how to take the combined object. `'text'` (default) parses
662
+ * accumulated assistant text. `'event'` reads an adapter-emitted
663
+ * `structured-output.complete` and does not parse prose.
663
664
  */
664
665
  finalStructuredOutput?: {
665
666
  jsonSchema: JSONSchema
@@ -667,6 +668,7 @@ interface TextEngineConfig<
667
668
  normalize?: (data: unknown) => unknown
668
669
  validate?: (data: unknown) => unknown
669
670
  nativeCombined?: boolean
671
+ source?: 'text' | 'event'
670
672
  }
671
673
  }
672
674
 
@@ -801,12 +803,14 @@ class TextEngine<
801
803
  code?: string
802
804
  cause?: unknown
803
805
  } | null = null
806
+ private combinedCompleteEmitted = false
804
807
  private readonly finalStructuredOutput?: {
805
808
  jsonSchema: JSONSchema
806
809
  yieldChunks: boolean
807
810
  normalize?: (data: unknown) => unknown
808
811
  validate?: (data: unknown) => unknown
809
812
  nativeCombined?: boolean
813
+ source?: 'text' | 'event'
810
814
  }
811
815
 
812
816
  constructor(
@@ -1110,7 +1114,8 @@ class TextEngine<
1110
1114
  this.finalStructuredOutput &&
1111
1115
  this.toolPhase !== 'wait' &&
1112
1116
  !this.isCancelled() &&
1113
- !this.finalizationError
1117
+ !this.finalizationError &&
1118
+ !this.earlyTermination
1114
1119
  ) {
1115
1120
  if (this.finalStructuredOutput.nativeCombined === true) {
1116
1121
  yield* this.harvestCombinedStructuredOutput()
@@ -1374,9 +1379,46 @@ class TextEngine<
1374
1379
  // text starting — intermediate tool-call iterations don't need it,
1375
1380
  // and emitting at run-start would wrap tool-call commentary into a
1376
1381
  // structured-output part too.
1382
+ if (
1383
+ chunk.type === EventType.CUSTOM &&
1384
+ chunk.name === 'structured-output.start'
1385
+ ) {
1386
+ this.combinedStartEmitted = true
1387
+ const startValue = chunk.value
1388
+ if (
1389
+ startValue &&
1390
+ typeof startValue === 'object' &&
1391
+ 'messageId' in startValue &&
1392
+ typeof startValue.messageId === 'string'
1393
+ ) {
1394
+ this.combinedStructuredMessageId = startValue.messageId
1395
+ }
1396
+ }
1397
+
1398
+ let outboundChunk: StreamChunk = chunk
1399
+ if (
1400
+ this.finalStructuredOutput?.source === 'event' &&
1401
+ chunk.type === EventType.CUSTOM &&
1402
+ chunk.name === 'structured-output.complete'
1403
+ ) {
1404
+ const parsed = readStructuredOutputCompleteValue(chunk.value)
1405
+ if (parsed) {
1406
+ const object = this.finalStructuredOutput.normalize
1407
+ ? this.finalStructuredOutput.normalize(parsed.object)
1408
+ : parsed.object
1409
+ this.structuredOutputResult = { data: object, rawText: parsed.raw }
1410
+ this.combinedCompleteEmitted = true
1411
+ const value = chunk.value
1412
+ if (object !== parsed.object && value && typeof value === 'object') {
1413
+ outboundChunk = { ...chunk, value: { ...value, object } }
1414
+ }
1415
+ }
1416
+ }
1417
+
1377
1418
  if (
1378
1419
  this.finalStructuredOutput?.nativeCombined === true &&
1379
1420
  this.finalStructuredOutput.yieldChunks &&
1421
+ this.finalStructuredOutput.source !== 'event' &&
1380
1422
  !this.combinedStartEmitted &&
1381
1423
  chunk.type === EventType.TEXT_MESSAGE_START
1382
1424
  ) {
@@ -1408,7 +1450,7 @@ class TextEngine<
1408
1450
  // Pipe chunk through middleware (devtools middleware observes; strip-to-spec cleans)
1409
1451
  const outputChunks = await this.middlewareRunner.runOnChunk(
1410
1452
  this.middlewareCtx,
1411
- chunk,
1453
+ outboundChunk,
1412
1454
  )
1413
1455
  // When a streaming structured-output finalization step will run after
1414
1456
  // the agent loop, suppress the agent-loop's RUN_STARTED/RUN_FINISHED
@@ -1554,9 +1596,23 @@ class TextEngine<
1554
1596
  }
1555
1597
 
1556
1598
  private handleRunErrorEvent(
1557
- _chunk: Extract<StreamChunk, { type: 'RUN_ERROR' }>,
1599
+ chunk: Extract<StreamChunk, { type: 'RUN_ERROR' }>,
1558
1600
  ): void {
1559
1601
  this.earlyTermination = true
1602
+ if (this.finalStructuredOutput && this.finalizationError === null) {
1603
+ const message =
1604
+ chunk.message ||
1605
+ chunk.error?.message ||
1606
+ 'Run failed before structured output completed'
1607
+ this.finalizationError = {
1608
+ message,
1609
+ ...(chunk.code !== undefined
1610
+ ? { code: chunk.code }
1611
+ : chunk.error?.code !== undefined
1612
+ ? { code: chunk.error.code }
1613
+ : {}),
1614
+ }
1615
+ }
1560
1616
  }
1561
1617
 
1562
1618
  private finalizeCurrentThinkingStep(): void {
@@ -2177,7 +2233,9 @@ class TextEngine<
2177
2233
  ? undefined
2178
2234
  : JSON.stringify(message.content)
2179
2235
  return {
2180
- id: `snapshot_${this.runIdOverride ?? this.requestId}_${index}`,
2236
+ id:
2237
+ message.id ||
2238
+ `snapshot_${this.runIdOverride ?? this.requestId}_${index}`,
2181
2239
  role: message.role,
2182
2240
  ...(content !== undefined ? { content } : {}),
2183
2241
  ...('toolCalls' in message && message.toolCalls
@@ -2855,7 +2913,9 @@ class TextEngine<
2855
2913
  return null
2856
2914
  }
2857
2915
 
2858
- const buildSynthesizedStart = (): StreamChunk => {
2916
+ // The synthetic event is inserted before its trigger, so share that
2917
+ // trigger's timestamp rather than making the earlier event sort later.
2918
+ const buildSynthesizedStart = (timestamp = Date.now()): StreamChunk => {
2859
2919
  const idForStart = structuredMessageId ?? generateMessageId()
2860
2920
  structuredMessageId = idForStart
2861
2921
  return {
@@ -2863,7 +2923,7 @@ class TextEngine<
2863
2923
  name: 'structured-output.start',
2864
2924
  value: { messageId: idForStart },
2865
2925
  model: this.params.model,
2866
- timestamp: Date.now(),
2926
+ timestamp,
2867
2927
  threadId: this.threadId,
2868
2928
  ...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
2869
2929
  }
@@ -2913,7 +2973,7 @@ class TextEngine<
2913
2973
  chunk.type === EventType.TEXT_MESSAGE_END)
2914
2974
  ) {
2915
2975
  startEmitted = true
2916
- const synthStart = buildSynthesizedStart()
2976
+ const synthStart = buildSynthesizedStart(chunk.timestamp)
2917
2977
  const synthOutputs = await pipeThroughMiddleware(synthStart)
2918
2978
  for (const outputChunk of synthOutputs) {
2919
2979
  yield outputChunk
@@ -2926,7 +2986,7 @@ class TextEngine<
2926
2986
  // of a silent UI.
2927
2987
  if (!startEmitted && chunk.type === EventType.RUN_ERROR) {
2928
2988
  startEmitted = true
2929
- const synthStart = buildSynthesizedStart()
2989
+ const synthStart = buildSynthesizedStart(chunk.timestamp)
2930
2990
  const synthOutputs = await pipeThroughMiddleware(synthStart)
2931
2991
  for (const outputChunk of synthOutputs) {
2932
2992
  yield outputChunk
@@ -3139,35 +3199,46 @@ class TextEngine<
3139
3199
  }
3140
3200
 
3141
3201
  const yieldChunks = this.finalStructuredOutput.yieldChunks
3142
- const rawText = this.accumulatedContent
3202
+ const source = this.finalStructuredOutput.source ?? 'text'
3143
3203
 
3144
- // Empty final-turn text means the agent loop terminated without the
3145
- // model emitting any assistant content (e.g. early termination after
3146
- // tool calls). Mirror the fallback path's "missing structured result"
3147
- // error rather than silently returning undefined.
3148
- if (rawText.length === 0) {
3149
- this.finalizationError = {
3150
- message: 'missing structured result',
3151
- code: 'structured-output-missing-result',
3204
+ if (source === 'event') {
3205
+ if (!this.structuredOutputResult) {
3206
+ this.finalizationError = {
3207
+ message: 'missing structured result',
3208
+ code: 'structured-output-missing-result',
3209
+ }
3152
3210
  }
3153
3211
  } else {
3154
- try {
3155
- const parsed: unknown = JSON.parse(rawText)
3156
- // Normalize (un-widen) before storing so the synthesized
3157
- // structured-output.complete chunk and the Promise<T> result both
3158
- // carry the cleaned payload. JSON.parse preserves provider nulls, so
3159
- // this is where native-combined output gets its widening undone.
3160
- const data = this.finalStructuredOutput.normalize
3161
- ? this.finalStructuredOutput.normalize(parsed)
3162
- : parsed
3163
- this.structuredOutputResult = { data, rawText }
3164
- } catch (err: unknown) {
3165
- const detail =
3166
- rawText.slice(0, 200) + (rawText.length > 200 ? '...' : '')
3212
+ const rawText = this.accumulatedContent
3213
+
3214
+ // Empty final-turn text means the agent loop terminated without the
3215
+ // model emitting any assistant content (e.g. early termination after
3216
+ // tool calls). Mirror the fallback path's "missing structured result"
3217
+ // error rather than silently returning undefined.
3218
+ if (rawText.length === 0) {
3167
3219
  this.finalizationError = {
3168
- message: `Failed to parse structured output as JSON. Content: ${detail}`,
3169
- code: 'structured-output-parse-failed',
3170
- cause: err,
3220
+ message: 'missing structured result',
3221
+ code: 'structured-output-missing-result',
3222
+ }
3223
+ } else {
3224
+ try {
3225
+ const parsed: unknown = JSON.parse(rawText)
3226
+ // Normalize (un-widen) before storing so the synthesized
3227
+ // structured-output.complete chunk and the Promise<T> result both
3228
+ // carry the cleaned payload. JSON.parse preserves provider nulls, so
3229
+ // this is where native-combined output gets its widening undone.
3230
+ const data = this.finalStructuredOutput.normalize
3231
+ ? this.finalStructuredOutput.normalize(parsed)
3232
+ : parsed
3233
+ this.structuredOutputResult = { data, rawText }
3234
+ } catch (err: unknown) {
3235
+ const detail =
3236
+ rawText.slice(0, 200) + (rawText.length > 200 ? '...' : '')
3237
+ this.finalizationError = {
3238
+ message: `Failed to parse structured output as JSON. Content: ${detail}`,
3239
+ code: 'structured-output-parse-failed',
3240
+ cause: err,
3241
+ }
3171
3242
  }
3172
3243
  }
3173
3244
  }
@@ -3237,7 +3308,11 @@ class TextEngine<
3237
3308
  // complete event yields AFTER the loop ends, by which point
3238
3309
  // `getActiveAssistantMessageId()` returns null and would otherwise drop
3239
3310
  // the event silently).
3240
- if (this.structuredOutputResult && !this.finalizationError) {
3311
+ if (
3312
+ this.structuredOutputResult &&
3313
+ !this.finalizationError &&
3314
+ !this.combinedCompleteEmitted
3315
+ ) {
3241
3316
  const completeChunk: StreamChunk = {
3242
3317
  type: EventType.CUSTOM,
3243
3318
  name: 'structured-output.complete',
@@ -3890,6 +3965,8 @@ async function runAgenticStructuredOutput<
3890
3965
  // agent loop's accumulated final-turn text.
3891
3966
  const nativeCombined =
3892
3967
  adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
3968
+ const source =
3969
+ adapter.combinedStructuredOutputSource?.(options.modelOptions) ?? 'text'
3893
3970
 
3894
3971
  const mcpManager = MCPManager.from(mcp)
3895
3972
  const mcpTools = await mcpManager.discover()
@@ -3913,6 +3990,7 @@ async function runAgenticStructuredOutput<
3913
3990
  normalize,
3914
3991
  ...(validate ? { validate } : {}),
3915
3992
  ...(nativeCombined ? { nativeCombined: true } : {}),
3993
+ source,
3916
3994
  },
3917
3995
  },
3918
3996
  logger,
@@ -4008,14 +4086,14 @@ async function* fallbackStructuredOutputStream(
4008
4086
  chatOptions.threadId ?? `fallback-${Date.now()}-${fallbackRand}`
4009
4087
  const messageId = `fallback-${Date.now()}-${fallbackRand}`
4010
4088
  const model = chatOptions.model
4011
- const timestamp = Date.now()
4089
+ const startedAt = Date.now()
4012
4090
 
4013
4091
  yield {
4014
4092
  type: EventType.RUN_STARTED,
4015
4093
  runId,
4016
4094
  threadId,
4017
4095
  model,
4018
- timestamp,
4096
+ timestamp: startedAt,
4019
4097
  }
4020
4098
 
4021
4099
  let result: StructuredOutputResult<unknown>
@@ -4029,7 +4107,7 @@ async function* fallbackStructuredOutputStream(
4029
4107
  runId,
4030
4108
  threadId,
4031
4109
  model,
4032
- timestamp,
4110
+ timestamp: Date.now(),
4033
4111
  message,
4034
4112
  error: { message },
4035
4113
  }
@@ -4041,7 +4119,7 @@ async function* fallbackStructuredOutputStream(
4041
4119
  messageId,
4042
4120
  role: 'assistant',
4043
4121
  model,
4044
- timestamp,
4122
+ timestamp: Date.now(),
4045
4123
  }
4046
4124
 
4047
4125
  yield {
@@ -4049,14 +4127,14 @@ async function* fallbackStructuredOutputStream(
4049
4127
  messageId,
4050
4128
  delta: result.rawText,
4051
4129
  model,
4052
- timestamp,
4130
+ timestamp: Date.now(),
4053
4131
  }
4054
4132
 
4055
4133
  yield {
4056
4134
  type: EventType.TEXT_MESSAGE_END,
4057
4135
  messageId,
4058
4136
  model,
4059
- timestamp,
4137
+ timestamp: Date.now(),
4060
4138
  }
4061
4139
 
4062
4140
  yield {
@@ -4064,7 +4142,7 @@ async function* fallbackStructuredOutputStream(
4064
4142
  name: 'structured-output.complete',
4065
4143
  value: { object: result.data, raw: result.rawText },
4066
4144
  model,
4067
- timestamp,
4145
+ timestamp: Date.now(),
4068
4146
  }
4069
4147
 
4070
4148
  yield {
@@ -4072,7 +4150,7 @@ async function* fallbackStructuredOutputStream(
4072
4150
  runId,
4073
4151
  threadId,
4074
4152
  model,
4075
- timestamp,
4153
+ timestamp: Date.now(),
4076
4154
  finishReason: 'stop',
4077
4155
  // Forward adapter-reported token usage so consumers reading
4078
4156
  // `RUN_FINISHED.usage` (and the engine's `runOnUsage` middleware hook) see
@@ -4205,6 +4283,8 @@ async function* runStreamingStructuredOutputImpl<
4205
4283
  // does not fire.
4206
4284
  const nativeCombined =
4207
4285
  adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
4286
+ const source =
4287
+ adapter.combinedStructuredOutputSource?.(options.modelOptions) ?? 'text'
4208
4288
 
4209
4289
  const mcpManager = MCPManager.from(mcp)
4210
4290
  const mcpTools = await mcpManager.discover()
@@ -4229,6 +4309,7 @@ async function* runStreamingStructuredOutputImpl<
4229
4309
  yieldChunks: true,
4230
4310
  normalize,
4231
4311
  ...(nativeCombined ? { nativeCombined: true } : {}),
4312
+ source,
4232
4313
  },
4233
4314
  },
4234
4315
  logger,