@tanstack/ai 0.21.3 → 0.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -312,11 +312,20 @@ interface TextEngineConfig<
312
312
  * as the validated result and retrievable via
313
313
  * `getValidatedStructuredOutput()`. Used by `runAgenticStructuredOutput`
314
314
  * to perform Standard Schema validation inside the engine.
315
+ * - nativeCombined: when true, the adapter declared
316
+ * `supportsCombinedToolsAndSchema()` and the engine wires `jsonSchema`
317
+ * into the regular `chatStream` call instead of running a separate
318
+ * finalization round-trip. The agent loop's final-turn text is the
319
+ * schema-constrained JSON; the engine parses it from accumulated
320
+ * content. The `'structuredOutput'` middleware phase does NOT fire on
321
+ * this path — middleware sees the run through `beforeModel` /
322
+ * `modelStream` as usual.
315
323
  */
316
324
  finalStructuredOutput?: {
317
325
  jsonSchema: JSONSchema
318
326
  yieldChunks: boolean
319
327
  validate?: (data: unknown) => unknown
328
+ nativeCombined?: boolean
320
329
  }
321
330
  }
322
331
 
@@ -379,6 +388,16 @@ class TextEngine<
379
388
  // Structured-output finalization state (populated by runStructuredFinalization)
380
389
  private structuredOutputResult: { data: unknown; rawText: string } | null =
381
390
  null
391
+ // Native combined mode: tracks whether we've already emitted the synthetic
392
+ // `structured-output.start` event before the schema-constrained final-turn
393
+ // text begins streaming. The event must precede the first
394
+ // TEXT_MESSAGE_START so the client-side StreamProcessor routes the JSON
395
+ // deltas into a StructuredOutputPart instead of a plain TextPart.
396
+ private combinedStartEmitted = false
397
+ // Native combined mode: messageId we want the synthetic
398
+ // `structured-output.start` (and any error emitted before deltas arrive)
399
+ // to carry, so the client matches it to the streaming text deltas.
400
+ private combinedStructuredMessageId: string | null = null
382
401
  // Holds the validated value when `finalStructuredOutput.validate` is provided
383
402
  // and succeeds. Distinct from `structuredOutputResult.data` (the raw,
384
403
  // unvalidated payload from the structured-output.complete chunk).
@@ -393,6 +412,7 @@ class TextEngine<
393
412
  jsonSchema: JSONSchema
394
413
  yieldChunks: boolean
395
414
  validate?: (data: unknown) => unknown
415
+ nativeCombined?: boolean
396
416
  }
397
417
 
398
418
  constructor(
@@ -461,6 +481,7 @@ class TextEngine<
461
481
  this.middlewareCtx = {
462
482
  requestId: this.requestId,
463
483
  streamId: this.streamId,
484
+ runId: this.runIdOverride ?? this.requestId,
464
485
  threadId: this.threadId,
465
486
  // Legacy alias kept on the ctx so middleware that reads
466
487
  // `ctx.conversationId` keeps working. Always equals `threadId`.
@@ -560,12 +581,19 @@ class TextEngine<
560
581
  return
561
582
  }
562
583
 
563
- // Skip the agent loop entirely when there are no tools AND a structured-
564
- // output finalization will run. Without tools the model has nothing to
565
- // do in the loop, so executing one iteration would burn an extra
566
- // provider call before the finalization request.
584
+ // Skip the agent loop entirely when there are no tools AND a separate
585
+ // structured-output finalization will run. Without tools the model has
586
+ // nothing to do in the loop, so executing one iteration would burn an
587
+ // extra provider call before the finalization request.
588
+ //
589
+ // Native combined mode does NOT skip — the agent loop itself produces
590
+ // the schema-constrained final answer in one pass (model emits the
591
+ // schema-constrained text on its natural final turn). Even with zero
592
+ // tools, the single chatStream call IS the structured-output call.
567
593
  const skipAgentLoop =
568
- !!this.finalStructuredOutput && this.tools.length === 0
594
+ !!this.finalStructuredOutput &&
595
+ this.tools.length === 0 &&
596
+ this.finalStructuredOutput.nativeCombined !== true
569
597
 
570
598
  if (!skipAgentLoop) {
571
599
  do {
@@ -584,11 +612,12 @@ class TextEngine<
584
612
  this.middlewareCtx.phase = 'beforeModel'
585
613
  this.middlewareCtx.iteration = this.iterationCount
586
614
  const iterConfig = this.buildMiddlewareConfig()
587
- const transformedConfig = await this.middlewareRunner.runOnConfig(
588
- this.middlewareCtx,
589
- iterConfig,
590
- )
591
- this.applyMiddlewareConfig(transformedConfig)
615
+ const iterTransformedConfig =
616
+ await this.middlewareRunner.runOnConfig(
617
+ this.middlewareCtx,
618
+ iterConfig,
619
+ )
620
+ this.applyMiddlewareConfig(iterTransformedConfig)
592
621
 
593
622
  yield* this.streamModelResponse()
594
623
  } else {
@@ -607,12 +636,20 @@ class TextEngine<
607
636
  // requested AND the run hasn't already errored/aborted, run it through
608
637
  // the middleware pipeline. The terminal hook fires once at the very
609
638
  // end (after finalization), not after the agent loop.
639
+ //
640
+ // Native combined mode takes a different path: the agent loop's final-
641
+ // turn text IS the schema-constrained JSON, so we harvest it from
642
+ // `accumulatedContent` instead of issuing a second provider call.
610
643
  if (
611
644
  this.finalStructuredOutput &&
612
645
  !this.isCancelled() &&
613
646
  !this.finalizationError
614
647
  ) {
615
- yield* this.runStructuredFinalization()
648
+ if (this.finalStructuredOutput.nativeCombined === true) {
649
+ yield* this.harvestCombinedStructuredOutput()
650
+ } else {
651
+ yield* this.runStructuredFinalization()
652
+ }
616
653
  }
617
654
 
618
655
  // Call terminal hook (skip when waiting for client — stream is paused, not finished).
@@ -777,6 +814,18 @@ class TextEngine<
777
814
  },
778
815
  )
779
816
 
817
+ // When the adapter declared `supportsCombinedToolsAndSchema()`, the
818
+ // activity layer set `nativeCombined: true` and we forward the
819
+ // pre-converted JSON Schema into the regular chatStream call. The
820
+ // adapter wires it into the upstream request (e.g. `response_format`,
821
+ // `text.format`, `output_format`) so the model's final-turn text is
822
+ // schema-constrained and the engine can harvest it from the agent loop
823
+ // without a separate finalization round-trip.
824
+ const combinedSchema =
825
+ this.finalStructuredOutput?.nativeCombined === true
826
+ ? this.finalStructuredOutput.jsonSchema
827
+ : undefined
828
+
780
829
  for await (const chunk of this.adapter.chatStream({
781
830
  model: this.params.model,
782
831
  messages: this.messages,
@@ -792,6 +841,7 @@ class TextEngine<
792
841
  threadId: this.threadId,
793
842
  runId: this.runIdOverride,
794
843
  parentRunId: this.parentRunIdOverride,
844
+ ...(combinedSchema ? { outputSchema: combinedSchema } : {}),
795
845
  })) {
796
846
  if (this.isCancelled()) {
797
847
  break
@@ -803,6 +853,44 @@ class TextEngine<
803
853
  // BEFORE middleware, so fields like finishReason, delta, etc. are available
804
854
  this.handleStreamChunk(chunk)
805
855
 
856
+ // Native combined mode: synthesize `structured-output.start` BEFORE
857
+ // the first TEXT_MESSAGE_START so the client-side StreamProcessor
858
+ // routes the schema-constrained JSON deltas into a
859
+ // StructuredOutputPart. We delay synthesis until we actually see
860
+ // text starting — intermediate tool-call iterations don't need it,
861
+ // and emitting at run-start would wrap tool-call commentary into a
862
+ // structured-output part too.
863
+ if (
864
+ this.finalStructuredOutput?.nativeCombined === true &&
865
+ this.finalStructuredOutput.yieldChunks &&
866
+ !this.combinedStartEmitted &&
867
+ chunk.type === EventType.TEXT_MESSAGE_START
868
+ ) {
869
+ this.combinedStartEmitted = true
870
+ const messageId =
871
+ typeof chunk.messageId === 'string' && chunk.messageId !== ''
872
+ ? chunk.messageId
873
+ : generateMessageId()
874
+ this.combinedStructuredMessageId = messageId
875
+ const synthStart: StreamChunk = {
876
+ type: EventType.CUSTOM,
877
+ name: 'structured-output.start',
878
+ value: { messageId },
879
+ model: this.params.model,
880
+ timestamp: Date.now(),
881
+ threadId: this.threadId,
882
+ ...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
883
+ }
884
+ const synthOutputs = await this.middlewareRunner.runOnChunk(
885
+ this.middlewareCtx,
886
+ synthStart,
887
+ )
888
+ for (const outputChunk of synthOutputs) {
889
+ yield outputChunk
890
+ this.middlewareCtx.chunkIndex++
891
+ }
892
+ }
893
+
806
894
  // Pipe chunk through middleware (devtools middleware observes; strip-to-spec cleans)
807
895
  const outputChunks = await this.middlewareRunner.runOnChunk(
808
896
  this.middlewareCtx,
@@ -812,8 +900,13 @@ class TextEngine<
812
900
  // the agent loop, suppress the agent-loop's RUN_STARTED/RUN_FINISHED
813
901
  // here — the finalization step emits the single outer lifecycle pair
814
902
  // that reaches the consumer.
903
+ //
904
+ // Native combined mode does NOT issue a second adapter stream — the
905
+ // agent loop's lifecycle IS the outer pair the consumer sees.
815
906
  const suppressAgentLifecycle =
816
- !!this.finalStructuredOutput && this.finalStructuredOutput.yieldChunks
907
+ !!this.finalStructuredOutput &&
908
+ this.finalStructuredOutput.yieldChunks &&
909
+ this.finalStructuredOutput.nativeCombined !== true
817
910
  for (const outputChunk of outputChunks) {
818
911
  if (
819
912
  suppressAgentLifecycle &&
@@ -1948,6 +2041,179 @@ class TextEngine<
1948
2041
  }
1949
2042
  }
1950
2043
 
2044
+ /**
2045
+ * Native combined mode: harvest the structured output from the agent
2046
+ * loop's accumulated final-turn text (no separate provider call).
2047
+ *
2048
+ * The adapter wired `outputSchema` into the regular `chatStream` request,
2049
+ * so the model's final-turn text is the schema-constrained JSON. We parse
2050
+ * `this.accumulatedContent`, populate `this.structuredOutputResult`, emit
2051
+ * a synthetic `structured-output.complete` (and a `structured-output.start`
2052
+ * if one wasn't emitted earlier — only happens on the streaming path when
2053
+ * the model returned no text at all), and run the validate callback when
2054
+ * present. Failures populate `this.finalizationError` so the engine's
2055
+ * terminal-hook chooser routes to `onError` (per spec §7.3).
2056
+ *
2057
+ * The `'structuredOutput'` middleware phase intentionally does NOT fire on
2058
+ * this path — middleware sees the run through `beforeModel` / `modelStream`
2059
+ * as usual. See PR #605 / issue #605 for the design rationale.
2060
+ */
2061
+ private async *harvestCombinedStructuredOutput(): AsyncGenerator<StreamChunk> {
2062
+ if (!this.finalStructuredOutput) {
2063
+ throw new Error(
2064
+ 'harvestCombinedStructuredOutput called without finalStructuredOutput config',
2065
+ )
2066
+ }
2067
+
2068
+ const yieldChunks = this.finalStructuredOutput.yieldChunks
2069
+ const rawText = this.accumulatedContent
2070
+
2071
+ // Empty final-turn text means the agent loop terminated without the
2072
+ // model emitting any assistant content (e.g. early termination after
2073
+ // tool calls). Mirror the fallback path's "missing structured result"
2074
+ // error rather than silently returning undefined.
2075
+ if (rawText.length === 0) {
2076
+ this.finalizationError = {
2077
+ message: 'missing structured result',
2078
+ code: 'structured-output-missing-result',
2079
+ }
2080
+ } else {
2081
+ try {
2082
+ const parsed: unknown = JSON.parse(rawText)
2083
+ this.structuredOutputResult = { data: parsed, rawText }
2084
+ } catch (err: unknown) {
2085
+ const detail =
2086
+ rawText.slice(0, 200) + (rawText.length > 200 ? '...' : '')
2087
+ this.finalizationError = {
2088
+ message: `Failed to parse structured output as JSON. Content: ${detail}`,
2089
+ code: 'structured-output-parse-failed',
2090
+ cause: err,
2091
+ }
2092
+ }
2093
+ }
2094
+
2095
+ // Validate against the Standard Schema (when supplied). Validation
2096
+ // failures route through onError just like the fallback path.
2097
+ if (
2098
+ this.structuredOutputResult &&
2099
+ !this.finalizationError &&
2100
+ this.finalStructuredOutput.validate
2101
+ ) {
2102
+ try {
2103
+ const validated = this.finalStructuredOutput.validate(
2104
+ this.structuredOutputResult.data,
2105
+ )
2106
+ this.validatedStructuredOutput = validated
2107
+ this.hasValidatedStructuredOutput = true
2108
+ } catch (err: unknown) {
2109
+ const message = err instanceof Error ? err.message : String(err)
2110
+ this.finalizationError = {
2111
+ message,
2112
+ code: 'structured-output-validation-failed',
2113
+ cause: err,
2114
+ }
2115
+ }
2116
+ }
2117
+
2118
+ if (!yieldChunks) {
2119
+ // Promise<T> path: state is populated, nothing to yield. The
2120
+ // activity-layer caller pulls `structuredOutputResult` /
2121
+ // `validatedStructuredOutput` directly.
2122
+ return
2123
+ }
2124
+
2125
+ // Streaming path: emit a synthetic `structured-output.start` if the
2126
+ // model produced no text at all (so the client snaps an errored
2127
+ // StructuredOutputPart rather than nothing). The normal path already
2128
+ // emitted start before the first TEXT_MESSAGE_START in
2129
+ // `streamModelResponse`.
2130
+ if (!this.combinedStartEmitted) {
2131
+ this.combinedStartEmitted = true
2132
+ const messageId = this.combinedStructuredMessageId ?? generateMessageId()
2133
+ this.combinedStructuredMessageId = messageId
2134
+ const synthStart: StreamChunk = {
2135
+ type: EventType.CUSTOM,
2136
+ name: 'structured-output.start',
2137
+ value: { messageId },
2138
+ model: this.params.model,
2139
+ timestamp: Date.now(),
2140
+ threadId: this.threadId,
2141
+ ...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
2142
+ }
2143
+ const startOutputs = await this.middlewareRunner.runOnChunk(
2144
+ this.middlewareCtx,
2145
+ synthStart,
2146
+ )
2147
+ for (const outputChunk of startOutputs) {
2148
+ yield outputChunk
2149
+ this.middlewareCtx.chunkIndex++
2150
+ }
2151
+ }
2152
+
2153
+ // On success, emit the synthetic `structured-output.complete` carrying
2154
+ // the parsed object + raw text. Pin the messageId so the client-side
2155
+ // handler can target the right UIMessage even when the agent loop's
2156
+ // terminal RUN_FINISHED has already cleared `activeMessageIds` (the
2157
+ // complete event yields AFTER the loop ends, by which point
2158
+ // `getActiveAssistantMessageId()` returns null and would otherwise drop
2159
+ // the event silently).
2160
+ if (this.structuredOutputResult && !this.finalizationError) {
2161
+ const completeChunk: StreamChunk = {
2162
+ type: EventType.CUSTOM,
2163
+ name: 'structured-output.complete',
2164
+ value: {
2165
+ object: this.structuredOutputResult.data,
2166
+ raw: this.structuredOutputResult.rawText,
2167
+ ...(this.combinedStructuredMessageId
2168
+ ? { messageId: this.combinedStructuredMessageId }
2169
+ : {}),
2170
+ },
2171
+ model: this.params.model,
2172
+ timestamp: Date.now(),
2173
+ threadId: this.threadId,
2174
+ ...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
2175
+ }
2176
+ const completeOutputs = await this.middlewareRunner.runOnChunk(
2177
+ this.middlewareCtx,
2178
+ completeChunk,
2179
+ )
2180
+ for (const outputChunk of completeOutputs) {
2181
+ yield outputChunk
2182
+ this.middlewareCtx.chunkIndex++
2183
+ }
2184
+ }
2185
+
2186
+ // On failure, emit a synthetic RUN_ERROR so the streaming consumer's
2187
+ // `for await` doesn't end silently. Mirrors the fallback path.
2188
+ if (this.finalizationError) {
2189
+ const errChunk: StreamChunk = {
2190
+ type: EventType.RUN_ERROR,
2191
+ runId: this.runIdOverride ?? this.requestId,
2192
+ model: this.params.model,
2193
+ timestamp: Date.now(),
2194
+ threadId: this.threadId,
2195
+ message: this.finalizationError.message,
2196
+ ...(this.finalizationError.code
2197
+ ? { code: this.finalizationError.code }
2198
+ : {}),
2199
+ error: {
2200
+ message: this.finalizationError.message,
2201
+ ...(this.finalizationError.code
2202
+ ? { code: this.finalizationError.code }
2203
+ : {}),
2204
+ },
2205
+ }
2206
+ const errOutputs = await this.middlewareRunner.runOnChunk(
2207
+ this.middlewareCtx,
2208
+ errChunk,
2209
+ )
2210
+ for (const outputChunk of errOutputs) {
2211
+ yield outputChunk
2212
+ this.middlewareCtx.chunkIndex++
2213
+ }
2214
+ }
2215
+ }
2216
+
1951
2217
  private buildMiddlewareConfig(): ChatMiddlewareConfig {
1952
2218
  return {
1953
2219
  messages: this.messages,
@@ -2243,6 +2509,13 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
2243
2509
  parseWithStandardSchema<InferSchemaType<TSchema>>(outputSchema, data)
2244
2510
  : undefined
2245
2511
 
2512
+ // Per issue #605: same capability check as the streaming path. When the
2513
+ // adapter handles tools + schema natively, the engine skips the separate
2514
+ // structured-output finalization call and harvests the JSON from the
2515
+ // agent loop's accumulated final-turn text.
2516
+ const nativeCombined =
2517
+ adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
2518
+
2246
2519
  const engine = new TextEngine(
2247
2520
  {
2248
2521
  adapter,
@@ -2256,6 +2529,7 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
2256
2529
  jsonSchema,
2257
2530
  yieldChunks: false,
2258
2531
  ...(validate ? { validate } : {}),
2532
+ ...(nativeCombined ? { nativeCombined: true } : {}),
2259
2533
  },
2260
2534
  },
2261
2535
  logger,
@@ -2493,6 +2767,16 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
2493
2767
  const model = adapter.model
2494
2768
  const logger = resolveDebugOption(debug)
2495
2769
 
2770
+ // Per issue #605: adapters that natively combine tools + schema-constrained
2771
+ // output in one streaming call (modern OpenAI, Anthropic 4.5+, Gemini 3+,
2772
+ // Grok 4+) opt in via `supportsCombinedToolsAndSchema()`. The engine then
2773
+ // forwards the schema into the regular `chatStream` call and harvests the
2774
+ // structured result from the agent loop's accumulated text — no separate
2775
+ // finalization round-trip, and the `'structuredOutput'` middleware phase
2776
+ // does not fire.
2777
+ const nativeCombined =
2778
+ adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
2779
+
2496
2780
  // Inputs may be UIMessages (from useChat) or ModelMessages (from server-side
2497
2781
  // callers). TextEngine handles the conversion uniformly.
2498
2782
  const engine = new TextEngine(
@@ -2504,7 +2788,11 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
2504
2788
  >,
2505
2789
  middleware,
2506
2790
  context,
2507
- finalStructuredOutput: { jsonSchema, yieldChunks: true },
2791
+ finalStructuredOutput: {
2792
+ jsonSchema,
2793
+ yieldChunks: true,
2794
+ ...(nativeCombined ? { nativeCombined: true } : {}),
2795
+ },
2508
2796
  },
2509
2797
  logger,
2510
2798
  )
@@ -38,6 +38,8 @@ export interface ChatMiddlewareContext {
38
38
  requestId: string
39
39
  /** Unique identifier for this stream */
40
40
  streamId: string
41
+ /** AG-UI run identifier for correlating client and server events */
42
+ runId: string
41
43
  /**
42
44
  * AG-UI thread identifier — a stable per-conversation ID used to
43
45
  * correlate client and server devtools events. Resolves to the
@@ -98,6 +98,17 @@ export interface StreamProcessorEvents {
98
98
  stepId: string,
99
99
  content: string,
100
100
  ) => void
101
+ onStructuredOutputChange?: (args: {
102
+ phase: 'start' | 'update' | 'complete' | 'error'
103
+ messageId: string
104
+ status: 'streaming' | 'complete' | 'error'
105
+ raw: string
106
+ partial?: unknown
107
+ data?: unknown
108
+ reasoning?: string
109
+ errorMessage?: string
110
+ delta?: string
111
+ }) => void
101
112
  }
102
113
 
103
114
  /**
@@ -116,6 +127,8 @@ export interface StreamProcessorOptions {
116
127
  initialMessages?: Array<UIMessage>
117
128
  }
118
129
 
130
+ const STRUCTURED_OUTPUT_UPDATE_BATCH_SIZE = 12
131
+
119
132
  /**
120
133
  * StreamProcessor - State machine for processing AI response streams
121
134
  *
@@ -149,6 +162,13 @@ export class StreamProcessor {
149
162
  private pendingThinkingStepId: string | null = null
150
163
 
151
164
  private readonly structuredMessageIds: Set<string> = new Set()
165
+ private readonly structuredOutputUpdateBatches = new Map<
166
+ string,
167
+ {
168
+ delta: string
169
+ chunkCount: number
170
+ }
171
+ >()
152
172
 
153
173
  // Run tracking (for concurrent run safety)
154
174
  private readonly activeRuns = new Set<string>()
@@ -401,6 +421,9 @@ export class StreamProcessor {
401
421
  for (const id of this.structuredMessageIds) {
402
422
  if (!keptIds.has(id)) this.structuredMessageIds.delete(id)
403
423
  }
424
+ for (const id of this.structuredOutputUpdateBatches.keys()) {
425
+ if (!keptIds.has(id)) this.structuredOutputUpdateBatches.delete(id)
426
+ }
404
427
  for (const id of this.messageStates.keys()) {
405
428
  if (!keptIds.has(id)) this.messageStates.delete(id)
406
429
  }
@@ -423,6 +446,7 @@ export class StreamProcessor {
423
446
  this.activeMessageIds.clear()
424
447
  this.toolCallToMessage.clear()
425
448
  this.structuredMessageIds.clear()
449
+ this.structuredOutputUpdateBatches.clear()
426
450
  this.pendingManualMessageId = null
427
451
  this.emitMessagesChange()
428
452
  }
@@ -899,6 +923,7 @@ export class StreamProcessor {
899
923
  delta,
900
924
  )
901
925
  state.totalTextContent += delta
926
+ this.queueStructuredOutputUpdate(messageId, delta)
902
927
  this.emitMessagesChange()
903
928
  }
904
929
  return
@@ -1269,12 +1294,14 @@ export class StreamProcessor {
1269
1294
  }
1270
1295
 
1271
1296
  if (this.structuredMessageIds.has(messageId)) {
1297
+ this.flushStructuredOutputUpdate(messageId)
1272
1298
  this.messages = errorStructuredOutputPart(
1273
1299
  this.messages,
1274
1300
  messageId,
1275
1301
  errorMessage,
1276
1302
  )
1277
1303
  this.structuredMessageIds.delete(messageId)
1304
+ this.emitStructuredOutputChange(messageId, 'error')
1278
1305
  this.emitMessagesChange()
1279
1306
  }
1280
1307
 
@@ -1463,6 +1490,13 @@ export class StreamProcessor {
1463
1490
  if (targetId) {
1464
1491
  this.ensureAssistantMessage(targetId)
1465
1492
  this.structuredMessageIds.add(targetId)
1493
+ this.structuredOutputUpdateBatches.delete(targetId)
1494
+ this.events.onStructuredOutputChange?.({
1495
+ phase: 'start',
1496
+ messageId: targetId,
1497
+ status: 'streaming',
1498
+ raw: '',
1499
+ })
1466
1500
  }
1467
1501
  return
1468
1502
  }
@@ -1476,6 +1510,7 @@ export class StreamProcessor {
1476
1510
  }
1477
1511
  const targetId = v.messageId ?? messageId
1478
1512
  if (targetId) {
1513
+ this.flushStructuredOutputUpdate(targetId)
1479
1514
  this.messages = completeStructuredOutputPart(
1480
1515
  this.messages,
1481
1516
  targetId,
@@ -1484,6 +1519,7 @@ export class StreamProcessor {
1484
1519
  v.reasoning,
1485
1520
  )
1486
1521
  this.structuredMessageIds.delete(targetId)
1522
+ this.emitStructuredOutputChange(targetId, 'complete')
1487
1523
  this.emitMessagesChange()
1488
1524
  }
1489
1525
  // Fall through so user `onCustomEvent` callbacks still observe the event.
@@ -1665,6 +1701,58 @@ export class StreamProcessor {
1665
1701
  this.events.onTextUpdate?.(messageId, state.currentSegmentText)
1666
1702
  }
1667
1703
 
1704
+ private queueStructuredOutputUpdate(messageId: string, delta: string): void {
1705
+ const existing = this.structuredOutputUpdateBatches.get(messageId)
1706
+ const next = {
1707
+ delta: `${existing?.delta ?? ''}${delta}`,
1708
+ chunkCount: (existing?.chunkCount ?? 0) + 1,
1709
+ }
1710
+
1711
+ this.structuredOutputUpdateBatches.set(messageId, next)
1712
+
1713
+ if (next.chunkCount >= STRUCTURED_OUTPUT_UPDATE_BATCH_SIZE) {
1714
+ this.flushStructuredOutputUpdate(messageId)
1715
+ }
1716
+ }
1717
+
1718
+ private flushStructuredOutputUpdate(messageId: string): void {
1719
+ const batch = this.structuredOutputUpdateBatches.get(messageId)
1720
+ if (!batch || batch.chunkCount === 0) return
1721
+
1722
+ this.structuredOutputUpdateBatches.delete(messageId)
1723
+ this.emitStructuredOutputChange(messageId, 'update', batch.delta)
1724
+ }
1725
+
1726
+ private emitStructuredOutputChange(
1727
+ messageId: string,
1728
+ phase: 'update' | 'complete' | 'error',
1729
+ delta?: string,
1730
+ ): void {
1731
+ const part = this.messages
1732
+ .find((message) => message.id === messageId)
1733
+ ?.parts.find(
1734
+ (
1735
+ messagePart,
1736
+ ): messagePart is Extract<MessagePart, { type: 'structured-output' }> =>
1737
+ messagePart.type === 'structured-output',
1738
+ )
1739
+ if (!part) return
1740
+
1741
+ this.events.onStructuredOutputChange?.({
1742
+ phase,
1743
+ messageId,
1744
+ status: part.status,
1745
+ raw: part.raw,
1746
+ ...(part.partial !== undefined ? { partial: part.partial } : {}),
1747
+ ...(part.data !== undefined ? { data: part.data } : {}),
1748
+ ...(part.reasoning !== undefined ? { reasoning: part.reasoning } : {}),
1749
+ ...(part.errorMessage !== undefined
1750
+ ? { errorMessage: part.errorMessage }
1751
+ : {}),
1752
+ ...(delta !== undefined ? { delta } : {}),
1753
+ })
1754
+ }
1755
+
1668
1756
  /**
1669
1757
  * Emit messages change event
1670
1758
  */
@@ -1718,13 +1806,16 @@ export class StreamProcessor {
1718
1806
  // definition a non-errored, never-completed run (the multi-run case:
1719
1807
  // run-A errors, run-B is still streaming when finalize fires).
1720
1808
  for (const messageId of this.structuredMessageIds) {
1809
+ this.flushStructuredOutputUpdate(messageId)
1721
1810
  this.messages = errorStructuredOutputPart(
1722
1811
  this.messages,
1723
1812
  messageId,
1724
1813
  'Stream ended without structured-output.complete',
1725
1814
  )
1815
+ this.emitStructuredOutputChange(messageId, 'error')
1726
1816
  }
1727
1817
  this.structuredMessageIds.clear()
1818
+ this.structuredOutputUpdateBatches.clear()
1728
1819
 
1729
1820
  this.activeMessageIds.clear()
1730
1821
 
@@ -1855,6 +1946,7 @@ export class StreamProcessor {
1855
1946
  this.activeRuns.clear()
1856
1947
  this.toolCallToMessage.clear()
1857
1948
  this.structuredMessageIds.clear()
1949
+ this.structuredOutputUpdateBatches.clear()
1858
1950
  this.pendingManualMessageId = null
1859
1951
  this.pendingThinkingStepId = null
1860
1952
  this.finishReason = null
@@ -103,10 +103,10 @@ function createId(prefix: string): string {
103
103
  * @example Generate speech from text
104
104
  * ```ts
105
105
  * import { generateSpeech } from '@tanstack/ai'
106
- * import { openaiTTS } from '@tanstack/ai-openai'
106
+ * import { openaiSpeech } from '@tanstack/ai-openai'
107
107
  *
108
108
  * const result = await generateSpeech({
109
- * adapter: openaiTTS('tts-1-hd'),
109
+ * adapter: openaiSpeech('tts-1-hd'),
110
110
  * text: 'Hello, welcome to TanStack AI!',
111
111
  * voice: 'nova'
112
112
  * })
@@ -117,7 +117,7 @@ function createId(prefix: string): string {
117
117
  * @example With format and speed options
118
118
  * ```ts
119
119
  * const result = await generateSpeech({
120
- * adapter: openaiTTS('tts-1'),
120
+ * adapter: openaiSpeech('tts-1'),
121
121
  * text: 'This is slower speech.',
122
122
  * voice: 'alloy',
123
123
  * format: 'wav',
package/src/types.ts CHANGED
@@ -800,10 +800,26 @@ export interface TextOptions<
800
800
 
801
801
  /**
802
802
  * Schema for structured output.
803
- * When provided, the adapter should use the provider's native structured output API
804
- * to ensure the response conforms to this schema.
805
- * The schema will be converted to JSON Schema format before being sent to the provider.
806
- * Supports any Standard JSON Schema compliant library (Zod, ArkType, Valibot, etc.).
803
+ *
804
+ * **Two distinct use sites:**
805
+ *
806
+ * 1. **User-facing (activity layer):** accepts any
807
+ * {@link SchemaInput} — Zod, ArkType, Valibot, or a raw JSON Schema.
808
+ * The activity layer converts to JSON Schema before handing off.
809
+ *
810
+ * 2. **Adapter-facing (`chatStream` call):** the engine populates this with
811
+ * a pre-converted JSON Schema **only** when the adapter declared
812
+ * `supportsCombinedToolsAndSchema(modelOptions) === true`. The adapter
813
+ * should then wire the schema into the upstream request (e.g.
814
+ * `response_format: { type: 'json_schema', ... }`, `text.format`,
815
+ * `output_format`) alongside any `tools`. The model's natural final
816
+ * turn carries the schema-constrained JSON text and the engine
817
+ * harvests it from the agent loop without a separate finalization
818
+ * round-trip.
819
+ *
820
+ * Adapters that did NOT declare the capability never see this field
821
+ * populated — the engine instead invokes `structuredOutput` /
822
+ * `structuredOutputStream` after the agent loop.
807
823
  */
808
824
  outputSchema?: SchemaInput
809
825
  /**