@tanstack/ai 0.21.3 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +20 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +185 -8
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/chat/middleware/types.d.ts +2 -0
- package/dist/esm/activities/chat/stream/processor.d.ts +15 -0
- package/dist/esm/activities/chat/stream/processor.js +56 -0
- package/dist/esm/activities/chat/stream/processor.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/types.d.ts +20 -4
- package/package.json +3 -3
- package/skills/ai-core/adapter-configuration/SKILL.md +32 -1
- package/skills/ai-core/debug-logging/SKILL.md +1 -1
- package/skills/ai-core/structured-outputs/SKILL.md +21 -9
- package/src/activities/chat/adapter.ts +23 -0
- package/src/activities/chat/index.ts +301 -13
- package/src/activities/chat/middleware/types.ts +2 -0
- package/src/activities/chat/stream/processor.ts +92 -0
- package/src/activities/generateSpeech/index.ts +3 -3
- package/src/types.ts +20 -4
|
@@ -312,11 +312,20 @@ interface TextEngineConfig<
|
|
|
312
312
|
* as the validated result and retrievable via
|
|
313
313
|
* `getValidatedStructuredOutput()`. Used by `runAgenticStructuredOutput`
|
|
314
314
|
* to perform Standard Schema validation inside the engine.
|
|
315
|
+
* - nativeCombined: when true, the adapter declared
|
|
316
|
+
* `supportsCombinedToolsAndSchema()` and the engine wires `jsonSchema`
|
|
317
|
+
* into the regular `chatStream` call instead of running a separate
|
|
318
|
+
* finalization round-trip. The agent loop's final-turn text is the
|
|
319
|
+
* schema-constrained JSON; the engine parses it from accumulated
|
|
320
|
+
* content. The `'structuredOutput'` middleware phase does NOT fire on
|
|
321
|
+
* this path — middleware sees the run through `beforeModel` /
|
|
322
|
+
* `modelStream` as usual.
|
|
315
323
|
*/
|
|
316
324
|
finalStructuredOutput?: {
|
|
317
325
|
jsonSchema: JSONSchema
|
|
318
326
|
yieldChunks: boolean
|
|
319
327
|
validate?: (data: unknown) => unknown
|
|
328
|
+
nativeCombined?: boolean
|
|
320
329
|
}
|
|
321
330
|
}
|
|
322
331
|
|
|
@@ -379,6 +388,16 @@ class TextEngine<
|
|
|
379
388
|
// Structured-output finalization state (populated by runStructuredFinalization)
|
|
380
389
|
private structuredOutputResult: { data: unknown; rawText: string } | null =
|
|
381
390
|
null
|
|
391
|
+
// Native combined mode: tracks whether we've already emitted the synthetic
|
|
392
|
+
// `structured-output.start` event before the schema-constrained final-turn
|
|
393
|
+
// text begins streaming. The event must precede the first
|
|
394
|
+
// TEXT_MESSAGE_START so the client-side StreamProcessor routes the JSON
|
|
395
|
+
// deltas into a StructuredOutputPart instead of a plain TextPart.
|
|
396
|
+
private combinedStartEmitted = false
|
|
397
|
+
// Native combined mode: messageId we want the synthetic
|
|
398
|
+
// `structured-output.start` (and any error emitted before deltas arrive)
|
|
399
|
+
// to carry, so the client matches it to the streaming text deltas.
|
|
400
|
+
private combinedStructuredMessageId: string | null = null
|
|
382
401
|
// Holds the validated value when `finalStructuredOutput.validate` is provided
|
|
383
402
|
// and succeeds. Distinct from `structuredOutputResult.data` (the raw,
|
|
384
403
|
// unvalidated payload from the structured-output.complete chunk).
|
|
@@ -393,6 +412,7 @@ class TextEngine<
|
|
|
393
412
|
jsonSchema: JSONSchema
|
|
394
413
|
yieldChunks: boolean
|
|
395
414
|
validate?: (data: unknown) => unknown
|
|
415
|
+
nativeCombined?: boolean
|
|
396
416
|
}
|
|
397
417
|
|
|
398
418
|
constructor(
|
|
@@ -461,6 +481,7 @@ class TextEngine<
|
|
|
461
481
|
this.middlewareCtx = {
|
|
462
482
|
requestId: this.requestId,
|
|
463
483
|
streamId: this.streamId,
|
|
484
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
464
485
|
threadId: this.threadId,
|
|
465
486
|
// Legacy alias kept on the ctx so middleware that reads
|
|
466
487
|
// `ctx.conversationId` keeps working. Always equals `threadId`.
|
|
@@ -560,12 +581,19 @@ class TextEngine<
|
|
|
560
581
|
return
|
|
561
582
|
}
|
|
562
583
|
|
|
563
|
-
// Skip the agent loop entirely when there are no tools AND a
|
|
564
|
-
// output finalization will run. Without tools the model has
|
|
565
|
-
// do in the loop, so executing one iteration would burn an
|
|
566
|
-
// provider call before the finalization request.
|
|
584
|
+
// Skip the agent loop entirely when there are no tools AND a separate
|
|
585
|
+
// structured-output finalization will run. Without tools the model has
|
|
586
|
+
// nothing to do in the loop, so executing one iteration would burn an
|
|
587
|
+
// extra provider call before the finalization request.
|
|
588
|
+
//
|
|
589
|
+
// Native combined mode does NOT skip — the agent loop itself produces
|
|
590
|
+
// the schema-constrained final answer in one pass (model emits the
|
|
591
|
+
// schema-constrained text on its natural final turn). Even with zero
|
|
592
|
+
// tools, the single chatStream call IS the structured-output call.
|
|
567
593
|
const skipAgentLoop =
|
|
568
|
-
!!this.finalStructuredOutput &&
|
|
594
|
+
!!this.finalStructuredOutput &&
|
|
595
|
+
this.tools.length === 0 &&
|
|
596
|
+
this.finalStructuredOutput.nativeCombined !== true
|
|
569
597
|
|
|
570
598
|
if (!skipAgentLoop) {
|
|
571
599
|
do {
|
|
@@ -584,11 +612,12 @@ class TextEngine<
|
|
|
584
612
|
this.middlewareCtx.phase = 'beforeModel'
|
|
585
613
|
this.middlewareCtx.iteration = this.iterationCount
|
|
586
614
|
const iterConfig = this.buildMiddlewareConfig()
|
|
587
|
-
const
|
|
588
|
-
this.
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
615
|
+
const iterTransformedConfig =
|
|
616
|
+
await this.middlewareRunner.runOnConfig(
|
|
617
|
+
this.middlewareCtx,
|
|
618
|
+
iterConfig,
|
|
619
|
+
)
|
|
620
|
+
this.applyMiddlewareConfig(iterTransformedConfig)
|
|
592
621
|
|
|
593
622
|
yield* this.streamModelResponse()
|
|
594
623
|
} else {
|
|
@@ -607,12 +636,20 @@ class TextEngine<
|
|
|
607
636
|
// requested AND the run hasn't already errored/aborted, run it through
|
|
608
637
|
// the middleware pipeline. The terminal hook fires once at the very
|
|
609
638
|
// end (after finalization), not after the agent loop.
|
|
639
|
+
//
|
|
640
|
+
// Native combined mode takes a different path: the agent loop's final-
|
|
641
|
+
// turn text IS the schema-constrained JSON, so we harvest it from
|
|
642
|
+
// `accumulatedContent` instead of issuing a second provider call.
|
|
610
643
|
if (
|
|
611
644
|
this.finalStructuredOutput &&
|
|
612
645
|
!this.isCancelled() &&
|
|
613
646
|
!this.finalizationError
|
|
614
647
|
) {
|
|
615
|
-
|
|
648
|
+
if (this.finalStructuredOutput.nativeCombined === true) {
|
|
649
|
+
yield* this.harvestCombinedStructuredOutput()
|
|
650
|
+
} else {
|
|
651
|
+
yield* this.runStructuredFinalization()
|
|
652
|
+
}
|
|
616
653
|
}
|
|
617
654
|
|
|
618
655
|
// Call terminal hook (skip when waiting for client — stream is paused, not finished).
|
|
@@ -777,6 +814,18 @@ class TextEngine<
|
|
|
777
814
|
},
|
|
778
815
|
)
|
|
779
816
|
|
|
817
|
+
// When the adapter declared `supportsCombinedToolsAndSchema()`, the
|
|
818
|
+
// activity layer set `nativeCombined: true` and we forward the
|
|
819
|
+
// pre-converted JSON Schema into the regular chatStream call. The
|
|
820
|
+
// adapter wires it into the upstream request (e.g. `response_format`,
|
|
821
|
+
// `text.format`, `output_format`) so the model's final-turn text is
|
|
822
|
+
// schema-constrained and the engine can harvest it from the agent loop
|
|
823
|
+
// without a separate finalization round-trip.
|
|
824
|
+
const combinedSchema =
|
|
825
|
+
this.finalStructuredOutput?.nativeCombined === true
|
|
826
|
+
? this.finalStructuredOutput.jsonSchema
|
|
827
|
+
: undefined
|
|
828
|
+
|
|
780
829
|
for await (const chunk of this.adapter.chatStream({
|
|
781
830
|
model: this.params.model,
|
|
782
831
|
messages: this.messages,
|
|
@@ -792,6 +841,7 @@ class TextEngine<
|
|
|
792
841
|
threadId: this.threadId,
|
|
793
842
|
runId: this.runIdOverride,
|
|
794
843
|
parentRunId: this.parentRunIdOverride,
|
|
844
|
+
...(combinedSchema ? { outputSchema: combinedSchema } : {}),
|
|
795
845
|
})) {
|
|
796
846
|
if (this.isCancelled()) {
|
|
797
847
|
break
|
|
@@ -803,6 +853,44 @@ class TextEngine<
|
|
|
803
853
|
// BEFORE middleware, so fields like finishReason, delta, etc. are available
|
|
804
854
|
this.handleStreamChunk(chunk)
|
|
805
855
|
|
|
856
|
+
// Native combined mode: synthesize `structured-output.start` BEFORE
|
|
857
|
+
// the first TEXT_MESSAGE_START so the client-side StreamProcessor
|
|
858
|
+
// routes the schema-constrained JSON deltas into a
|
|
859
|
+
// StructuredOutputPart. We delay synthesis until we actually see
|
|
860
|
+
// text starting — intermediate tool-call iterations don't need it,
|
|
861
|
+
// and emitting at run-start would wrap tool-call commentary into a
|
|
862
|
+
// structured-output part too.
|
|
863
|
+
if (
|
|
864
|
+
this.finalStructuredOutput?.nativeCombined === true &&
|
|
865
|
+
this.finalStructuredOutput.yieldChunks &&
|
|
866
|
+
!this.combinedStartEmitted &&
|
|
867
|
+
chunk.type === EventType.TEXT_MESSAGE_START
|
|
868
|
+
) {
|
|
869
|
+
this.combinedStartEmitted = true
|
|
870
|
+
const messageId =
|
|
871
|
+
typeof chunk.messageId === 'string' && chunk.messageId !== ''
|
|
872
|
+
? chunk.messageId
|
|
873
|
+
: generateMessageId()
|
|
874
|
+
this.combinedStructuredMessageId = messageId
|
|
875
|
+
const synthStart: StreamChunk = {
|
|
876
|
+
type: EventType.CUSTOM,
|
|
877
|
+
name: 'structured-output.start',
|
|
878
|
+
value: { messageId },
|
|
879
|
+
model: this.params.model,
|
|
880
|
+
timestamp: Date.now(),
|
|
881
|
+
threadId: this.threadId,
|
|
882
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
883
|
+
}
|
|
884
|
+
const synthOutputs = await this.middlewareRunner.runOnChunk(
|
|
885
|
+
this.middlewareCtx,
|
|
886
|
+
synthStart,
|
|
887
|
+
)
|
|
888
|
+
for (const outputChunk of synthOutputs) {
|
|
889
|
+
yield outputChunk
|
|
890
|
+
this.middlewareCtx.chunkIndex++
|
|
891
|
+
}
|
|
892
|
+
}
|
|
893
|
+
|
|
806
894
|
// Pipe chunk through middleware (devtools middleware observes; strip-to-spec cleans)
|
|
807
895
|
const outputChunks = await this.middlewareRunner.runOnChunk(
|
|
808
896
|
this.middlewareCtx,
|
|
@@ -812,8 +900,13 @@ class TextEngine<
|
|
|
812
900
|
// the agent loop, suppress the agent-loop's RUN_STARTED/RUN_FINISHED
|
|
813
901
|
// here — the finalization step emits the single outer lifecycle pair
|
|
814
902
|
// that reaches the consumer.
|
|
903
|
+
//
|
|
904
|
+
// Native combined mode does NOT issue a second adapter stream — the
|
|
905
|
+
// agent loop's lifecycle IS the outer pair the consumer sees.
|
|
815
906
|
const suppressAgentLifecycle =
|
|
816
|
-
!!this.finalStructuredOutput &&
|
|
907
|
+
!!this.finalStructuredOutput &&
|
|
908
|
+
this.finalStructuredOutput.yieldChunks &&
|
|
909
|
+
this.finalStructuredOutput.nativeCombined !== true
|
|
817
910
|
for (const outputChunk of outputChunks) {
|
|
818
911
|
if (
|
|
819
912
|
suppressAgentLifecycle &&
|
|
@@ -1948,6 +2041,179 @@ class TextEngine<
|
|
|
1948
2041
|
}
|
|
1949
2042
|
}
|
|
1950
2043
|
|
|
2044
|
+
/**
|
|
2045
|
+
* Native combined mode: harvest the structured output from the agent
|
|
2046
|
+
* loop's accumulated final-turn text (no separate provider call).
|
|
2047
|
+
*
|
|
2048
|
+
* The adapter wired `outputSchema` into the regular `chatStream` request,
|
|
2049
|
+
* so the model's final-turn text is the schema-constrained JSON. We parse
|
|
2050
|
+
* `this.accumulatedContent`, populate `this.structuredOutputResult`, emit
|
|
2051
|
+
* a synthetic `structured-output.complete` (and a `structured-output.start`
|
|
2052
|
+
* if one wasn't emitted earlier — only happens on the streaming path when
|
|
2053
|
+
* the model returned no text at all), and run the validate callback when
|
|
2054
|
+
* present. Failures populate `this.finalizationError` so the engine's
|
|
2055
|
+
* terminal-hook chooser routes to `onError` (per spec §7.3).
|
|
2056
|
+
*
|
|
2057
|
+
* The `'structuredOutput'` middleware phase intentionally does NOT fire on
|
|
2058
|
+
* this path — middleware sees the run through `beforeModel` / `modelStream`
|
|
2059
|
+
* as usual. See PR #605 / issue #605 for the design rationale.
|
|
2060
|
+
*/
|
|
2061
|
+
private async *harvestCombinedStructuredOutput(): AsyncGenerator<StreamChunk> {
|
|
2062
|
+
if (!this.finalStructuredOutput) {
|
|
2063
|
+
throw new Error(
|
|
2064
|
+
'harvestCombinedStructuredOutput called without finalStructuredOutput config',
|
|
2065
|
+
)
|
|
2066
|
+
}
|
|
2067
|
+
|
|
2068
|
+
const yieldChunks = this.finalStructuredOutput.yieldChunks
|
|
2069
|
+
const rawText = this.accumulatedContent
|
|
2070
|
+
|
|
2071
|
+
// Empty final-turn text means the agent loop terminated without the
|
|
2072
|
+
// model emitting any assistant content (e.g. early termination after
|
|
2073
|
+
// tool calls). Mirror the fallback path's "missing structured result"
|
|
2074
|
+
// error rather than silently returning undefined.
|
|
2075
|
+
if (rawText.length === 0) {
|
|
2076
|
+
this.finalizationError = {
|
|
2077
|
+
message: 'missing structured result',
|
|
2078
|
+
code: 'structured-output-missing-result',
|
|
2079
|
+
}
|
|
2080
|
+
} else {
|
|
2081
|
+
try {
|
|
2082
|
+
const parsed: unknown = JSON.parse(rawText)
|
|
2083
|
+
this.structuredOutputResult = { data: parsed, rawText }
|
|
2084
|
+
} catch (err: unknown) {
|
|
2085
|
+
const detail =
|
|
2086
|
+
rawText.slice(0, 200) + (rawText.length > 200 ? '...' : '')
|
|
2087
|
+
this.finalizationError = {
|
|
2088
|
+
message: `Failed to parse structured output as JSON. Content: ${detail}`,
|
|
2089
|
+
code: 'structured-output-parse-failed',
|
|
2090
|
+
cause: err,
|
|
2091
|
+
}
|
|
2092
|
+
}
|
|
2093
|
+
}
|
|
2094
|
+
|
|
2095
|
+
// Validate against the Standard Schema (when supplied). Validation
|
|
2096
|
+
// failures route through onError just like the fallback path.
|
|
2097
|
+
if (
|
|
2098
|
+
this.structuredOutputResult &&
|
|
2099
|
+
!this.finalizationError &&
|
|
2100
|
+
this.finalStructuredOutput.validate
|
|
2101
|
+
) {
|
|
2102
|
+
try {
|
|
2103
|
+
const validated = this.finalStructuredOutput.validate(
|
|
2104
|
+
this.structuredOutputResult.data,
|
|
2105
|
+
)
|
|
2106
|
+
this.validatedStructuredOutput = validated
|
|
2107
|
+
this.hasValidatedStructuredOutput = true
|
|
2108
|
+
} catch (err: unknown) {
|
|
2109
|
+
const message = err instanceof Error ? err.message : String(err)
|
|
2110
|
+
this.finalizationError = {
|
|
2111
|
+
message,
|
|
2112
|
+
code: 'structured-output-validation-failed',
|
|
2113
|
+
cause: err,
|
|
2114
|
+
}
|
|
2115
|
+
}
|
|
2116
|
+
}
|
|
2117
|
+
|
|
2118
|
+
if (!yieldChunks) {
|
|
2119
|
+
// Promise<T> path: state is populated, nothing to yield. The
|
|
2120
|
+
// activity-layer caller pulls `structuredOutputResult` /
|
|
2121
|
+
// `validatedStructuredOutput` directly.
|
|
2122
|
+
return
|
|
2123
|
+
}
|
|
2124
|
+
|
|
2125
|
+
// Streaming path: emit a synthetic `structured-output.start` if the
|
|
2126
|
+
// model produced no text at all (so the client snaps an errored
|
|
2127
|
+
// StructuredOutputPart rather than nothing). The normal path already
|
|
2128
|
+
// emitted start before the first TEXT_MESSAGE_START in
|
|
2129
|
+
// `streamModelResponse`.
|
|
2130
|
+
if (!this.combinedStartEmitted) {
|
|
2131
|
+
this.combinedStartEmitted = true
|
|
2132
|
+
const messageId = this.combinedStructuredMessageId ?? generateMessageId()
|
|
2133
|
+
this.combinedStructuredMessageId = messageId
|
|
2134
|
+
const synthStart: StreamChunk = {
|
|
2135
|
+
type: EventType.CUSTOM,
|
|
2136
|
+
name: 'structured-output.start',
|
|
2137
|
+
value: { messageId },
|
|
2138
|
+
model: this.params.model,
|
|
2139
|
+
timestamp: Date.now(),
|
|
2140
|
+
threadId: this.threadId,
|
|
2141
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
2142
|
+
}
|
|
2143
|
+
const startOutputs = await this.middlewareRunner.runOnChunk(
|
|
2144
|
+
this.middlewareCtx,
|
|
2145
|
+
synthStart,
|
|
2146
|
+
)
|
|
2147
|
+
for (const outputChunk of startOutputs) {
|
|
2148
|
+
yield outputChunk
|
|
2149
|
+
this.middlewareCtx.chunkIndex++
|
|
2150
|
+
}
|
|
2151
|
+
}
|
|
2152
|
+
|
|
2153
|
+
// On success, emit the synthetic `structured-output.complete` carrying
|
|
2154
|
+
// the parsed object + raw text. Pin the messageId so the client-side
|
|
2155
|
+
// handler can target the right UIMessage even when the agent loop's
|
|
2156
|
+
// terminal RUN_FINISHED has already cleared `activeMessageIds` (the
|
|
2157
|
+
// complete event yields AFTER the loop ends, by which point
|
|
2158
|
+
// `getActiveAssistantMessageId()` returns null and would otherwise drop
|
|
2159
|
+
// the event silently).
|
|
2160
|
+
if (this.structuredOutputResult && !this.finalizationError) {
|
|
2161
|
+
const completeChunk: StreamChunk = {
|
|
2162
|
+
type: EventType.CUSTOM,
|
|
2163
|
+
name: 'structured-output.complete',
|
|
2164
|
+
value: {
|
|
2165
|
+
object: this.structuredOutputResult.data,
|
|
2166
|
+
raw: this.structuredOutputResult.rawText,
|
|
2167
|
+
...(this.combinedStructuredMessageId
|
|
2168
|
+
? { messageId: this.combinedStructuredMessageId }
|
|
2169
|
+
: {}),
|
|
2170
|
+
},
|
|
2171
|
+
model: this.params.model,
|
|
2172
|
+
timestamp: Date.now(),
|
|
2173
|
+
threadId: this.threadId,
|
|
2174
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
2175
|
+
}
|
|
2176
|
+
const completeOutputs = await this.middlewareRunner.runOnChunk(
|
|
2177
|
+
this.middlewareCtx,
|
|
2178
|
+
completeChunk,
|
|
2179
|
+
)
|
|
2180
|
+
for (const outputChunk of completeOutputs) {
|
|
2181
|
+
yield outputChunk
|
|
2182
|
+
this.middlewareCtx.chunkIndex++
|
|
2183
|
+
}
|
|
2184
|
+
}
|
|
2185
|
+
|
|
2186
|
+
// On failure, emit a synthetic RUN_ERROR so the streaming consumer's
|
|
2187
|
+
// `for await` doesn't end silently. Mirrors the fallback path.
|
|
2188
|
+
if (this.finalizationError) {
|
|
2189
|
+
const errChunk: StreamChunk = {
|
|
2190
|
+
type: EventType.RUN_ERROR,
|
|
2191
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
2192
|
+
model: this.params.model,
|
|
2193
|
+
timestamp: Date.now(),
|
|
2194
|
+
threadId: this.threadId,
|
|
2195
|
+
message: this.finalizationError.message,
|
|
2196
|
+
...(this.finalizationError.code
|
|
2197
|
+
? { code: this.finalizationError.code }
|
|
2198
|
+
: {}),
|
|
2199
|
+
error: {
|
|
2200
|
+
message: this.finalizationError.message,
|
|
2201
|
+
...(this.finalizationError.code
|
|
2202
|
+
? { code: this.finalizationError.code }
|
|
2203
|
+
: {}),
|
|
2204
|
+
},
|
|
2205
|
+
}
|
|
2206
|
+
const errOutputs = await this.middlewareRunner.runOnChunk(
|
|
2207
|
+
this.middlewareCtx,
|
|
2208
|
+
errChunk,
|
|
2209
|
+
)
|
|
2210
|
+
for (const outputChunk of errOutputs) {
|
|
2211
|
+
yield outputChunk
|
|
2212
|
+
this.middlewareCtx.chunkIndex++
|
|
2213
|
+
}
|
|
2214
|
+
}
|
|
2215
|
+
}
|
|
2216
|
+
|
|
1951
2217
|
private buildMiddlewareConfig(): ChatMiddlewareConfig {
|
|
1952
2218
|
return {
|
|
1953
2219
|
messages: this.messages,
|
|
@@ -2243,6 +2509,13 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
|
|
|
2243
2509
|
parseWithStandardSchema<InferSchemaType<TSchema>>(outputSchema, data)
|
|
2244
2510
|
: undefined
|
|
2245
2511
|
|
|
2512
|
+
// Per issue #605: same capability check as the streaming path. When the
|
|
2513
|
+
// adapter handles tools + schema natively, the engine skips the separate
|
|
2514
|
+
// structured-output finalization call and harvests the JSON from the
|
|
2515
|
+
// agent loop's accumulated final-turn text.
|
|
2516
|
+
const nativeCombined =
|
|
2517
|
+
adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
|
|
2518
|
+
|
|
2246
2519
|
const engine = new TextEngine(
|
|
2247
2520
|
{
|
|
2248
2521
|
adapter,
|
|
@@ -2256,6 +2529,7 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
|
|
|
2256
2529
|
jsonSchema,
|
|
2257
2530
|
yieldChunks: false,
|
|
2258
2531
|
...(validate ? { validate } : {}),
|
|
2532
|
+
...(nativeCombined ? { nativeCombined: true } : {}),
|
|
2259
2533
|
},
|
|
2260
2534
|
},
|
|
2261
2535
|
logger,
|
|
@@ -2493,6 +2767,16 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
|
|
|
2493
2767
|
const model = adapter.model
|
|
2494
2768
|
const logger = resolveDebugOption(debug)
|
|
2495
2769
|
|
|
2770
|
+
// Per issue #605: adapters that natively combine tools + schema-constrained
|
|
2771
|
+
// output in one streaming call (modern OpenAI, Anthropic 4.5+, Gemini 3+,
|
|
2772
|
+
// Grok 4+) opt in via `supportsCombinedToolsAndSchema()`. The engine then
|
|
2773
|
+
// forwards the schema into the regular `chatStream` call and harvests the
|
|
2774
|
+
// structured result from the agent loop's accumulated text — no separate
|
|
2775
|
+
// finalization round-trip, and the `'structuredOutput'` middleware phase
|
|
2776
|
+
// does not fire.
|
|
2777
|
+
const nativeCombined =
|
|
2778
|
+
adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
|
|
2779
|
+
|
|
2496
2780
|
// Inputs may be UIMessages (from useChat) or ModelMessages (from server-side
|
|
2497
2781
|
// callers). TextEngine handles the conversion uniformly.
|
|
2498
2782
|
const engine = new TextEngine(
|
|
@@ -2504,7 +2788,11 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
|
|
|
2504
2788
|
>,
|
|
2505
2789
|
middleware,
|
|
2506
2790
|
context,
|
|
2507
|
-
finalStructuredOutput: {
|
|
2791
|
+
finalStructuredOutput: {
|
|
2792
|
+
jsonSchema,
|
|
2793
|
+
yieldChunks: true,
|
|
2794
|
+
...(nativeCombined ? { nativeCombined: true } : {}),
|
|
2795
|
+
},
|
|
2508
2796
|
},
|
|
2509
2797
|
logger,
|
|
2510
2798
|
)
|
|
@@ -38,6 +38,8 @@ export interface ChatMiddlewareContext {
|
|
|
38
38
|
requestId: string
|
|
39
39
|
/** Unique identifier for this stream */
|
|
40
40
|
streamId: string
|
|
41
|
+
/** AG-UI run identifier for correlating client and server events */
|
|
42
|
+
runId: string
|
|
41
43
|
/**
|
|
42
44
|
* AG-UI thread identifier — a stable per-conversation ID used to
|
|
43
45
|
* correlate client and server devtools events. Resolves to the
|
|
@@ -98,6 +98,17 @@ export interface StreamProcessorEvents {
|
|
|
98
98
|
stepId: string,
|
|
99
99
|
content: string,
|
|
100
100
|
) => void
|
|
101
|
+
onStructuredOutputChange?: (args: {
|
|
102
|
+
phase: 'start' | 'update' | 'complete' | 'error'
|
|
103
|
+
messageId: string
|
|
104
|
+
status: 'streaming' | 'complete' | 'error'
|
|
105
|
+
raw: string
|
|
106
|
+
partial?: unknown
|
|
107
|
+
data?: unknown
|
|
108
|
+
reasoning?: string
|
|
109
|
+
errorMessage?: string
|
|
110
|
+
delta?: string
|
|
111
|
+
}) => void
|
|
101
112
|
}
|
|
102
113
|
|
|
103
114
|
/**
|
|
@@ -116,6 +127,8 @@ export interface StreamProcessorOptions {
|
|
|
116
127
|
initialMessages?: Array<UIMessage>
|
|
117
128
|
}
|
|
118
129
|
|
|
130
|
+
const STRUCTURED_OUTPUT_UPDATE_BATCH_SIZE = 12
|
|
131
|
+
|
|
119
132
|
/**
|
|
120
133
|
* StreamProcessor - State machine for processing AI response streams
|
|
121
134
|
*
|
|
@@ -149,6 +162,13 @@ export class StreamProcessor {
|
|
|
149
162
|
private pendingThinkingStepId: string | null = null
|
|
150
163
|
|
|
151
164
|
private readonly structuredMessageIds: Set<string> = new Set()
|
|
165
|
+
private readonly structuredOutputUpdateBatches = new Map<
|
|
166
|
+
string,
|
|
167
|
+
{
|
|
168
|
+
delta: string
|
|
169
|
+
chunkCount: number
|
|
170
|
+
}
|
|
171
|
+
>()
|
|
152
172
|
|
|
153
173
|
// Run tracking (for concurrent run safety)
|
|
154
174
|
private readonly activeRuns = new Set<string>()
|
|
@@ -401,6 +421,9 @@ export class StreamProcessor {
|
|
|
401
421
|
for (const id of this.structuredMessageIds) {
|
|
402
422
|
if (!keptIds.has(id)) this.structuredMessageIds.delete(id)
|
|
403
423
|
}
|
|
424
|
+
for (const id of this.structuredOutputUpdateBatches.keys()) {
|
|
425
|
+
if (!keptIds.has(id)) this.structuredOutputUpdateBatches.delete(id)
|
|
426
|
+
}
|
|
404
427
|
for (const id of this.messageStates.keys()) {
|
|
405
428
|
if (!keptIds.has(id)) this.messageStates.delete(id)
|
|
406
429
|
}
|
|
@@ -423,6 +446,7 @@ export class StreamProcessor {
|
|
|
423
446
|
this.activeMessageIds.clear()
|
|
424
447
|
this.toolCallToMessage.clear()
|
|
425
448
|
this.structuredMessageIds.clear()
|
|
449
|
+
this.structuredOutputUpdateBatches.clear()
|
|
426
450
|
this.pendingManualMessageId = null
|
|
427
451
|
this.emitMessagesChange()
|
|
428
452
|
}
|
|
@@ -899,6 +923,7 @@ export class StreamProcessor {
|
|
|
899
923
|
delta,
|
|
900
924
|
)
|
|
901
925
|
state.totalTextContent += delta
|
|
926
|
+
this.queueStructuredOutputUpdate(messageId, delta)
|
|
902
927
|
this.emitMessagesChange()
|
|
903
928
|
}
|
|
904
929
|
return
|
|
@@ -1269,12 +1294,14 @@ export class StreamProcessor {
|
|
|
1269
1294
|
}
|
|
1270
1295
|
|
|
1271
1296
|
if (this.structuredMessageIds.has(messageId)) {
|
|
1297
|
+
this.flushStructuredOutputUpdate(messageId)
|
|
1272
1298
|
this.messages = errorStructuredOutputPart(
|
|
1273
1299
|
this.messages,
|
|
1274
1300
|
messageId,
|
|
1275
1301
|
errorMessage,
|
|
1276
1302
|
)
|
|
1277
1303
|
this.structuredMessageIds.delete(messageId)
|
|
1304
|
+
this.emitStructuredOutputChange(messageId, 'error')
|
|
1278
1305
|
this.emitMessagesChange()
|
|
1279
1306
|
}
|
|
1280
1307
|
|
|
@@ -1463,6 +1490,13 @@ export class StreamProcessor {
|
|
|
1463
1490
|
if (targetId) {
|
|
1464
1491
|
this.ensureAssistantMessage(targetId)
|
|
1465
1492
|
this.structuredMessageIds.add(targetId)
|
|
1493
|
+
this.structuredOutputUpdateBatches.delete(targetId)
|
|
1494
|
+
this.events.onStructuredOutputChange?.({
|
|
1495
|
+
phase: 'start',
|
|
1496
|
+
messageId: targetId,
|
|
1497
|
+
status: 'streaming',
|
|
1498
|
+
raw: '',
|
|
1499
|
+
})
|
|
1466
1500
|
}
|
|
1467
1501
|
return
|
|
1468
1502
|
}
|
|
@@ -1476,6 +1510,7 @@ export class StreamProcessor {
|
|
|
1476
1510
|
}
|
|
1477
1511
|
const targetId = v.messageId ?? messageId
|
|
1478
1512
|
if (targetId) {
|
|
1513
|
+
this.flushStructuredOutputUpdate(targetId)
|
|
1479
1514
|
this.messages = completeStructuredOutputPart(
|
|
1480
1515
|
this.messages,
|
|
1481
1516
|
targetId,
|
|
@@ -1484,6 +1519,7 @@ export class StreamProcessor {
|
|
|
1484
1519
|
v.reasoning,
|
|
1485
1520
|
)
|
|
1486
1521
|
this.structuredMessageIds.delete(targetId)
|
|
1522
|
+
this.emitStructuredOutputChange(targetId, 'complete')
|
|
1487
1523
|
this.emitMessagesChange()
|
|
1488
1524
|
}
|
|
1489
1525
|
// Fall through so user `onCustomEvent` callbacks still observe the event.
|
|
@@ -1665,6 +1701,58 @@ export class StreamProcessor {
|
|
|
1665
1701
|
this.events.onTextUpdate?.(messageId, state.currentSegmentText)
|
|
1666
1702
|
}
|
|
1667
1703
|
|
|
1704
|
+
private queueStructuredOutputUpdate(messageId: string, delta: string): void {
|
|
1705
|
+
const existing = this.structuredOutputUpdateBatches.get(messageId)
|
|
1706
|
+
const next = {
|
|
1707
|
+
delta: `${existing?.delta ?? ''}${delta}`,
|
|
1708
|
+
chunkCount: (existing?.chunkCount ?? 0) + 1,
|
|
1709
|
+
}
|
|
1710
|
+
|
|
1711
|
+
this.structuredOutputUpdateBatches.set(messageId, next)
|
|
1712
|
+
|
|
1713
|
+
if (next.chunkCount >= STRUCTURED_OUTPUT_UPDATE_BATCH_SIZE) {
|
|
1714
|
+
this.flushStructuredOutputUpdate(messageId)
|
|
1715
|
+
}
|
|
1716
|
+
}
|
|
1717
|
+
|
|
1718
|
+
private flushStructuredOutputUpdate(messageId: string): void {
|
|
1719
|
+
const batch = this.structuredOutputUpdateBatches.get(messageId)
|
|
1720
|
+
if (!batch || batch.chunkCount === 0) return
|
|
1721
|
+
|
|
1722
|
+
this.structuredOutputUpdateBatches.delete(messageId)
|
|
1723
|
+
this.emitStructuredOutputChange(messageId, 'update', batch.delta)
|
|
1724
|
+
}
|
|
1725
|
+
|
|
1726
|
+
private emitStructuredOutputChange(
|
|
1727
|
+
messageId: string,
|
|
1728
|
+
phase: 'update' | 'complete' | 'error',
|
|
1729
|
+
delta?: string,
|
|
1730
|
+
): void {
|
|
1731
|
+
const part = this.messages
|
|
1732
|
+
.find((message) => message.id === messageId)
|
|
1733
|
+
?.parts.find(
|
|
1734
|
+
(
|
|
1735
|
+
messagePart,
|
|
1736
|
+
): messagePart is Extract<MessagePart, { type: 'structured-output' }> =>
|
|
1737
|
+
messagePart.type === 'structured-output',
|
|
1738
|
+
)
|
|
1739
|
+
if (!part) return
|
|
1740
|
+
|
|
1741
|
+
this.events.onStructuredOutputChange?.({
|
|
1742
|
+
phase,
|
|
1743
|
+
messageId,
|
|
1744
|
+
status: part.status,
|
|
1745
|
+
raw: part.raw,
|
|
1746
|
+
...(part.partial !== undefined ? { partial: part.partial } : {}),
|
|
1747
|
+
...(part.data !== undefined ? { data: part.data } : {}),
|
|
1748
|
+
...(part.reasoning !== undefined ? { reasoning: part.reasoning } : {}),
|
|
1749
|
+
...(part.errorMessage !== undefined
|
|
1750
|
+
? { errorMessage: part.errorMessage }
|
|
1751
|
+
: {}),
|
|
1752
|
+
...(delta !== undefined ? { delta } : {}),
|
|
1753
|
+
})
|
|
1754
|
+
}
|
|
1755
|
+
|
|
1668
1756
|
/**
|
|
1669
1757
|
* Emit messages change event
|
|
1670
1758
|
*/
|
|
@@ -1718,13 +1806,16 @@ export class StreamProcessor {
|
|
|
1718
1806
|
// definition a non-errored, never-completed run (the multi-run case:
|
|
1719
1807
|
// run-A errors, run-B is still streaming when finalize fires).
|
|
1720
1808
|
for (const messageId of this.structuredMessageIds) {
|
|
1809
|
+
this.flushStructuredOutputUpdate(messageId)
|
|
1721
1810
|
this.messages = errorStructuredOutputPart(
|
|
1722
1811
|
this.messages,
|
|
1723
1812
|
messageId,
|
|
1724
1813
|
'Stream ended without structured-output.complete',
|
|
1725
1814
|
)
|
|
1815
|
+
this.emitStructuredOutputChange(messageId, 'error')
|
|
1726
1816
|
}
|
|
1727
1817
|
this.structuredMessageIds.clear()
|
|
1818
|
+
this.structuredOutputUpdateBatches.clear()
|
|
1728
1819
|
|
|
1729
1820
|
this.activeMessageIds.clear()
|
|
1730
1821
|
|
|
@@ -1855,6 +1946,7 @@ export class StreamProcessor {
|
|
|
1855
1946
|
this.activeRuns.clear()
|
|
1856
1947
|
this.toolCallToMessage.clear()
|
|
1857
1948
|
this.structuredMessageIds.clear()
|
|
1949
|
+
this.structuredOutputUpdateBatches.clear()
|
|
1858
1950
|
this.pendingManualMessageId = null
|
|
1859
1951
|
this.pendingThinkingStepId = null
|
|
1860
1952
|
this.finishReason = null
|
|
@@ -103,10 +103,10 @@ function createId(prefix: string): string {
|
|
|
103
103
|
* @example Generate speech from text
|
|
104
104
|
* ```ts
|
|
105
105
|
* import { generateSpeech } from '@tanstack/ai'
|
|
106
|
-
* import {
|
|
106
|
+
* import { openaiSpeech } from '@tanstack/ai-openai'
|
|
107
107
|
*
|
|
108
108
|
* const result = await generateSpeech({
|
|
109
|
-
* adapter:
|
|
109
|
+
* adapter: openaiSpeech('tts-1-hd'),
|
|
110
110
|
* text: 'Hello, welcome to TanStack AI!',
|
|
111
111
|
* voice: 'nova'
|
|
112
112
|
* })
|
|
@@ -117,7 +117,7 @@ function createId(prefix: string): string {
|
|
|
117
117
|
* @example With format and speed options
|
|
118
118
|
* ```ts
|
|
119
119
|
* const result = await generateSpeech({
|
|
120
|
-
* adapter:
|
|
120
|
+
* adapter: openaiSpeech('tts-1'),
|
|
121
121
|
* text: 'This is slower speech.',
|
|
122
122
|
* voice: 'alloy',
|
|
123
123
|
* format: 'wav',
|
package/src/types.ts
CHANGED
|
@@ -800,10 +800,26 @@ export interface TextOptions<
|
|
|
800
800
|
|
|
801
801
|
/**
|
|
802
802
|
* Schema for structured output.
|
|
803
|
-
*
|
|
804
|
-
*
|
|
805
|
-
*
|
|
806
|
-
*
|
|
803
|
+
*
|
|
804
|
+
* **Two distinct use sites:**
|
|
805
|
+
*
|
|
806
|
+
* 1. **User-facing (activity layer):** accepts any
|
|
807
|
+
* {@link SchemaInput} — Zod, ArkType, Valibot, or a raw JSON Schema.
|
|
808
|
+
* The activity layer converts to JSON Schema before handing off.
|
|
809
|
+
*
|
|
810
|
+
* 2. **Adapter-facing (`chatStream` call):** the engine populates this with
|
|
811
|
+
* a pre-converted JSON Schema **only** when the adapter declared
|
|
812
|
+
* `supportsCombinedToolsAndSchema(modelOptions) === true`. The adapter
|
|
813
|
+
* should then wire the schema into the upstream request (e.g.
|
|
814
|
+
* `response_format: { type: 'json_schema', ... }`, `text.format`,
|
|
815
|
+
* `output_format`) alongside any `tools`. The model's natural final
|
|
816
|
+
* turn carries the schema-constrained JSON text and the engine
|
|
817
|
+
* harvests it from the agent loop without a separate finalization
|
|
818
|
+
* round-trip.
|
|
819
|
+
*
|
|
820
|
+
* Adapters that did NOT declare the capability never see this field
|
|
821
|
+
* populated — the engine instead invokes `structuredOutput` /
|
|
822
|
+
* `structuredOutputStream` after the agent loop.
|
|
807
823
|
*/
|
|
808
824
|
outputSchema?: SchemaInput
|
|
809
825
|
/**
|