@tanstack/ai 0.21.3 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/activities/chat/adapter.d.ts +20 -0
- package/dist/esm/activities/chat/adapter.js.map +1 -1
- package/dist/esm/activities/chat/index.js +184 -8
- package/dist/esm/activities/chat/index.js.map +1 -1
- package/dist/esm/activities/generateSpeech/index.d.ts +3 -3
- package/dist/esm/activities/generateSpeech/index.js.map +1 -1
- package/dist/esm/types.d.ts +20 -4
- package/package.json +2 -2
- package/skills/ai-core/adapter-configuration/SKILL.md +31 -0
- package/skills/ai-core/structured-outputs/SKILL.md +21 -9
- package/src/activities/chat/adapter.ts +23 -0
- package/src/activities/chat/index.ts +300 -13
- package/src/activities/generateSpeech/index.ts +3 -3
- package/src/types.ts +20 -4
|
@@ -58,10 +58,10 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
|
|
|
58
58
|
* @example Generate speech from text
|
|
59
59
|
* ```ts
|
|
60
60
|
* import { generateSpeech } from '@tanstack/ai'
|
|
61
|
-
* import {
|
|
61
|
+
* import { openaiSpeech } from '@tanstack/ai-openai'
|
|
62
62
|
*
|
|
63
63
|
* const result = await generateSpeech({
|
|
64
|
-
* adapter:
|
|
64
|
+
* adapter: openaiSpeech('tts-1-hd'),
|
|
65
65
|
* text: 'Hello, welcome to TanStack AI!',
|
|
66
66
|
* voice: 'nova'
|
|
67
67
|
* })
|
|
@@ -72,7 +72,7 @@ export type TTSActivityResult<TStream extends boolean = false> = TStream extends
|
|
|
72
72
|
* @example With format and speed options
|
|
73
73
|
* ```ts
|
|
74
74
|
* const result = await generateSpeech({
|
|
75
|
-
* adapter:
|
|
75
|
+
* adapter: openaiSpeech('tts-1'),
|
|
76
76
|
* text: 'This is slower speech.',
|
|
77
77
|
* voice: 'alloy',
|
|
78
78
|
* format: 'wav',
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import {
|
|
1
|
+
{"version":3,"file":"index.js","sources":["../../../../src/activities/generateSpeech/index.ts"],"sourcesContent":["/**\n * TTS Activity\n *\n * Generates speech audio from text using text-to-speech models.\n * This is a self-contained module with implementation, types, and JSDoc.\n */\n\nimport { aiEventClient } from '@tanstack/ai-event-client'\nimport { streamGenerationResult } from '../stream-generation-result.js'\nimport { resolveDebugOption } from '../../logger/resolve'\nimport type { InternalLogger } from '../../logger/internal-logger'\nimport type { DebugOption } from '../../logger/types'\nimport type { TTSAdapter } from './adapter'\nimport type { StreamChunk, TTSResult } from '../../types'\n\n// ===========================\n// Activity Kind\n// ===========================\n\n/** The adapter kind this activity handles */\nexport const kind = 'tts' as const\n\n// ===========================\n// Type Extraction Helpers\n// ===========================\n\n/**\n * Extract provider options from a TTSAdapter via ~types.\n */\nexport type TTSProviderOptions<TAdapter> =\n TAdapter extends TTSAdapter<any, any>\n ? TAdapter['~types']['providerOptions']\n : object\n\n// ===========================\n// Activity Options Type\n// ===========================\n\n/**\n * Options for the TTS activity.\n * The model is extracted from the adapter's model property.\n *\n * @template TAdapter - The TTS adapter type\n * @template TStream - Whether to stream the output\n */\nexport interface TTSActivityOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n> {\n /** The TTS adapter to use (must be created with a model) */\n adapter: TAdapter & { kind: typeof kind }\n /** The text to convert to speech */\n text: string\n /** The voice to use for generation */\n voice?: string\n /** The output audio format */\n format?: 'mp3' | 'opus' | 'aac' | 'flac' | 'wav' | 'pcm'\n /** The speed of the generated audio (0.25 to 4.0) */\n speed?: number\n /** Provider-specific options for TTS generation */\n modelOptions?: TTSProviderOptions<TAdapter>\n /**\n * Whether to stream the generation result.\n * When true, returns an AsyncIterable<StreamChunk> for streaming transport.\n * When false or not provided, returns a Promise<TTSResult>.\n *\n * @default false\n */\n stream?: TStream\n /**\n * Enable debug logging. Pass `true` to enable all categories, `false` to\n * silence everything including errors, or a `DebugConfig` object for granular\n * control and/or a custom `Logger`.\n */\n debug?: DebugOption\n}\n\n// ===========================\n// Activity Result Type\n// ===========================\n\n/**\n * Result type for the TTS activity.\n * - If stream is true: AsyncIterable<StreamChunk>\n * - Otherwise: Promise<TTSResult>\n */\nexport type TTSActivityResult<TStream extends boolean = false> =\n TStream extends true ? AsyncIterable<StreamChunk> : Promise<TTSResult>\n\nfunction createId(prefix: string): string {\n return `${prefix}-${Date.now()}-${Math.random().toString(36).slice(2, 9)}`\n}\n\n// ===========================\n// Activity Implementation\n// ===========================\n\n/**\n * TTS activity - generates speech from text.\n *\n * Uses AI text-to-speech models to create audio from natural language text.\n *\n * @example Generate speech from text\n * ```ts\n * import { generateSpeech } from '@tanstack/ai'\n * import { openaiSpeech } from '@tanstack/ai-openai'\n *\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1-hd'),\n * text: 'Hello, welcome to TanStack AI!',\n * voice: 'nova'\n * })\n *\n * console.log(result.audio) // base64-encoded audio\n * ```\n *\n * @example With format and speed options\n * ```ts\n * const result = await generateSpeech({\n * adapter: openaiSpeech('tts-1'),\n * text: 'This is slower speech.',\n * voice: 'alloy',\n * format: 'wav',\n * speed: 0.8\n * })\n * ```\n */\nexport function generateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(options: TTSActivityOptions<TAdapter, TStream>): TTSActivityResult<TStream> {\n if (options.stream) {\n return streamGenerationResult(() =>\n runGenerateSpeech(options),\n ) as TTSActivityResult<TStream>\n }\n return runGenerateSpeech(options) as TTSActivityResult<TStream>\n}\n\n/**\n * Run the core TTS generation logic (non-streaming).\n */\nasync function runGenerateSpeech<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n>(options: TTSActivityOptions<TAdapter, boolean>): Promise<TTSResult> {\n const { adapter, stream: _stream, debug: _debug, ...rest } = options\n const model = adapter.model\n const requestId = createId('speech')\n const startTime = Date.now()\n const logger: InternalLogger = resolveDebugOption(options.debug)\n const providerName =\n (adapter as { name?: string; provider?: string }).provider ??\n (adapter as { name?: string }).name ??\n 'unknown'\n\n aiEventClient.emit('speech:request:started', {\n requestId,\n provider: adapter.name,\n model,\n text: rest.text,\n voice: rest.voice,\n format: rest.format,\n speed: rest.speed,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: startTime,\n })\n\n logger.request(`activity=generateSpeech provider=${providerName}`, {\n provider: providerName,\n model,\n })\n\n try {\n const result = await adapter.generateSpeech({ ...rest, model, logger })\n const duration = Date.now() - startTime\n\n aiEventClient.emit('speech:request:completed', {\n requestId,\n provider: adapter.name,\n model,\n audio: result.audio,\n format: result.format,\n audioDuration: result.duration,\n contentType: result.contentType,\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n\n logger.output(`activity=generateSpeech bytes=${result.audio.length}`, {\n bytes: result.audio.length,\n contentType: result.contentType,\n })\n\n return result\n } catch (error) {\n const duration = Date.now() - startTime\n const err = error as Error\n aiEventClient.emit('speech:request:error', {\n requestId,\n provider: adapter.name,\n model,\n error: { message: err.message, name: err.name },\n duration,\n modelOptions: rest.modelOptions as Record<string, unknown> | undefined,\n timestamp: Date.now(),\n })\n logger.errors('generateSpeech activity failed', {\n error,\n source: 'generateSpeech',\n })\n throw error\n }\n}\n\n// ===========================\n// Options Factory\n// ===========================\n\n/**\n * Create typed options for the generateSpeech() function without executing.\n */\nexport function createSpeechOptions<\n TAdapter extends TTSAdapter<string, TTSProviderOptions<TAdapter>>,\n TStream extends boolean = false,\n>(\n options: TTSActivityOptions<TAdapter, TStream>,\n): TTSActivityOptions<TAdapter, TStream> {\n return options\n}\n\n// Re-export adapter types\nexport type { TTSAdapter, TTSAdapterConfig, AnyTTSAdapter } from './adapter'\nexport { BaseTTSAdapter } from './adapter'\n"],"names":[],"mappings":";;;AAoBO,MAAM,OAAO;AAqEpB,SAAS,SAAS,QAAwB;AACxC,SAAO,GAAG,MAAM,IAAI,KAAK,IAAA,CAAK,IAAI,KAAK,OAAA,EAAS,SAAS,EAAE,EAAE,MAAM,GAAG,CAAC,CAAC;AAC1E;AAoCO,SAAS,eAGd,SAA4E;AAC5E,MAAI,QAAQ,QAAQ;AAClB,WAAO;AAAA,MAAuB,MAC5B,kBAAkB,OAAO;AAAA,IAAA;AAAA,EAE7B;AACA,SAAO,kBAAkB,OAAO;AAClC;AAKA,eAAe,kBAEb,SAAoE;AACpE,QAAM,EAAE,SAAS,QAAQ,SAAS,OAAO,QAAQ,GAAG,SAAS;AAC7D,QAAM,QAAQ,QAAQ;AACtB,QAAM,YAAY,SAAS,QAAQ;AACnC,QAAM,YAAY,KAAK,IAAA;AACvB,QAAM,SAAyB,mBAAmB,QAAQ,KAAK;AAC/D,QAAM,eACH,QAAiD,YACjD,QAA8B,QAC/B;AAEF,gBAAc,KAAK,0BAA0B;AAAA,IAC3C;AAAA,IACA,UAAU,QAAQ;AAAA,IAClB;AAAA,IACA,MAAM,KAAK;AAAA,IACX,OAAO,KAAK;AAAA,IACZ,QAAQ,KAAK;AAAA,IACb,OAAO,KAAK;AAAA,IACZ,cAAc,KAAK;AAAA,IACnB,WAAW;AAAA,EAAA,CACZ;AAED,SAAO,QAAQ,oCAAoC,YAAY,IAAI;AAAA,IACjE,UAAU;AAAA,IACV;AAAA,EAAA,CACD;AAED,MAAI;AACF,UAAM,SAAS,MAAM,QAAQ,eAAe,EAAE,GAAG,MAAM,OAAO,QAAQ;AACtE,UAAM,WAAW,KAAK,IAAA,IAAQ;AAE9B,kBAAc,KAAK,4BAA4B;AAAA,MAC7C;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,OAAO;AAAA,MACd,QAAQ,OAAO;AAAA,MACf,eAAe,OAAO;AAAA,MACtB,aAAa,OAAO;AAAA,MACpB;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AAED,WAAO,OAAO,iCAAiC,OAAO,MAAM,MAAM,IAAI;AAAA,MACpE,OAAO,OAAO,MAAM;AAAA,MACpB,aAAa,OAAO;AAAA,IAAA,CACrB;AAED,WAAO;AAAA,EACT,SAAS,OAAO;AACd,UAAM,WAAW,KAAK,IAAA,IAAQ;AAC9B,UAAM,MAAM;AACZ,kBAAc,KAAK,wBAAwB;AAAA,MACzC;AAAA,MACA,UAAU,QAAQ;AAAA,MAClB;AAAA,MACA,OAAO,EAAE,SAAS,IAAI,SAAS,MAAM,IAAI,KAAA;AAAA,MACzC;AAAA,MACA,cAAc,KAAK;AAAA,MACnB,WAAW,KAAK,IAAA;AAAA,IAAI,CACrB;AACD,WAAO,OAAO,kCAAkC;AAAA,MAC9C;AAAA,MACA,QAAQ;AAAA,IAAA,CACT;AACD,UAAM;AAAA,EACR;AACF;AASO,SAAS,oBAId,SACuC;AACvC,SAAO;AACT;"}
|
package/dist/esm/types.d.ts
CHANGED
|
@@ -650,10 +650,26 @@ export interface TextOptions<TProviderOptionsSuperset extends Record<string, any
|
|
|
650
650
|
request?: Request | RequestInit;
|
|
651
651
|
/**
|
|
652
652
|
* Schema for structured output.
|
|
653
|
-
*
|
|
654
|
-
*
|
|
655
|
-
*
|
|
656
|
-
*
|
|
653
|
+
*
|
|
654
|
+
* **Two distinct use sites:**
|
|
655
|
+
*
|
|
656
|
+
* 1. **User-facing (activity layer):** accepts any
|
|
657
|
+
* {@link SchemaInput} — Zod, ArkType, Valibot, or a raw JSON Schema.
|
|
658
|
+
* The activity layer converts to JSON Schema before handing off.
|
|
659
|
+
*
|
|
660
|
+
* 2. **Adapter-facing (`chatStream` call):** the engine populates this with
|
|
661
|
+
* a pre-converted JSON Schema **only** when the adapter declared
|
|
662
|
+
* `supportsCombinedToolsAndSchema(modelOptions) === true`. The adapter
|
|
663
|
+
* should then wire the schema into the upstream request (e.g.
|
|
664
|
+
* `response_format: { type: 'json_schema', ... }`, `text.format`,
|
|
665
|
+
* `output_format`) alongside any `tools`. The model's natural final
|
|
666
|
+
* turn carries the schema-constrained JSON text and the engine
|
|
667
|
+
* harvests it from the agent loop without a separate finalization
|
|
668
|
+
* round-trip.
|
|
669
|
+
*
|
|
670
|
+
* Adapters that did NOT declare the capability never see this field
|
|
671
|
+
* populated — the engine instead invokes `structuredOutput` /
|
|
672
|
+
* `structuredOutputStream` after the agent loop.
|
|
657
673
|
*/
|
|
658
674
|
outputSchema?: SchemaInput;
|
|
659
675
|
/**
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tanstack/ai",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.22.0",
|
|
4
4
|
"description": "Type-safe TypeScript AI SDK for streaming chat, tool calling, agents, structured outputs, and multimodal generation.",
|
|
5
5
|
"author": "Tanner Linsley",
|
|
6
6
|
"license": "MIT",
|
|
@@ -64,7 +64,7 @@
|
|
|
64
64
|
"@ag-ui/core": "^0.0.52",
|
|
65
65
|
"@standard-schema/spec": "^1.1.0",
|
|
66
66
|
"partial-json": "^0.1.7",
|
|
67
|
-
"@tanstack/ai-event-client": "0.3.
|
|
67
|
+
"@tanstack/ai-event-client": "0.3.11"
|
|
68
68
|
},
|
|
69
69
|
"peerDependencies": {
|
|
70
70
|
"@opentelemetry/api": ">=1.9.0"
|
|
@@ -221,6 +221,37 @@ const custom = myOpenai('ft:gpt-5.2:my-org:custom-model:abc123')
|
|
|
221
221
|
At runtime, `extendAdapter` simply passes through to the original factory.
|
|
222
222
|
The `_customModels` parameter is only used for type inference.
|
|
223
223
|
|
|
224
|
+
### 5. Capability Flag: `supportsCombinedToolsAndSchema`
|
|
225
|
+
|
|
226
|
+
Adapters can declare an optional capability method:
|
|
227
|
+
|
|
228
|
+
```ts
|
|
229
|
+
supportsCombinedToolsAndSchema?(modelOptions?: TProviderOptions): boolean
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
When `true`, the engine wires `outputSchema` into the regular
|
|
233
|
+
`chatStream` call alongside `tools` and harvests the schema-constrained
|
|
234
|
+
JSON from the agent loop's final-turn text — skipping the separate
|
|
235
|
+
`structuredOutput` / `structuredOutputStream` finalization round-trip.
|
|
236
|
+
When `false` (or the method is omitted), the legacy finalization path
|
|
237
|
+
runs.
|
|
238
|
+
|
|
239
|
+
Current per-adapter status (#605):
|
|
240
|
+
|
|
241
|
+
| Adapter | Returns |
|
|
242
|
+
| -------------------------------------------- | ------------------------------------------------------------------------------------------------- |
|
|
243
|
+
| `openaiText` / `openaiChatCompletions` | `true` (all supported models) |
|
|
244
|
+
| `anthropicText` | `true` for Claude 4.5+ (gated by `ANTHROPIC_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
245
|
+
| `geminiText` | `true` for Gemini 3.x (gated by `GEMINI_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
246
|
+
| `grokText` | `true` for Grok 4 family (gated by `GROK_COMBINED_TOOLS_AND_SCHEMA_MODELS`), `false` otherwise |
|
|
247
|
+
| `groqText` | `false` (Groq API rejects schema + tools + stream) |
|
|
248
|
+
| `openRouterText` / `openRouterResponsesText` | `false` (per-call resolution is a follow-up) |
|
|
249
|
+
| `ollamaText` | `false` (constrained-decoding vs tool-call grammar conflict) |
|
|
250
|
+
|
|
251
|
+
Subclasses can override to narrow the capability. When extending an
|
|
252
|
+
adapter for a custom model that doesn't support the combination, return
|
|
253
|
+
`false` explicitly.
|
|
254
|
+
|
|
224
255
|
## Common Mistakes
|
|
225
256
|
|
|
226
257
|
### a. HIGH: Confusing legacy monolithic with tree-shakeable adapter
|
|
@@ -181,15 +181,27 @@ The terminal event is a `CUSTOM` chunk: `{ type: 'CUSTOM', name: 'structured-out
|
|
|
181
181
|
|
|
182
182
|
**Adapter coverage for streaming:**
|
|
183
183
|
|
|
184
|
-
| Adapter
|
|
185
|
-
|
|
|
186
|
-
| `@tanstack/ai-openai`
|
|
187
|
-
| `@tanstack/ai-
|
|
188
|
-
| `@tanstack/ai-
|
|
189
|
-
| `@tanstack/ai-
|
|
190
|
-
|
|
|
191
|
-
|
|
192
|
-
|
|
184
|
+
| Adapter | `outputSchema` + `stream: true` |
|
|
185
|
+
| --------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
186
|
+
| `@tanstack/ai-openai` (Responses + Chat Completions) | **Native combined mode (#605)** — schema wired into the regular `chatStream` call alongside `tools`; engine harvests JSON, no finalization round-trip |
|
|
187
|
+
| `@tanstack/ai-anthropic` (Claude 4.5+ only) | **Native combined mode (#605)** — `output_config.format` + `tools` in one beta Messages call. Older Claude models fall back |
|
|
188
|
+
| `@tanstack/ai-gemini` (Gemini 3.x only) | **Native combined mode (#605)** — `responseSchema` + `tools` in one `generateContentStream`. Gemini 2.x falls back |
|
|
189
|
+
| `@tanstack/ai-grok` (Grok 4 family only) | **Native combined mode (#605)** — `response_format: json_schema` + `tools`. Grok 2 / 3 fall back |
|
|
190
|
+
| `@tanstack/ai-openrouter` | Native single-request stream (legacy `structuredOutputStream` path; per-call combined-mode lookup is a follow-up) |
|
|
191
|
+
| `@tanstack/ai-groq` | Legacy `structuredOutputStream` only (no tools — Groq's API rejects schema + tools + stream) |
|
|
192
|
+
| All other adapters (ollama, older Claude, Gemini 2.x, Grok 2/3) | Fallback: runs non-streaming `structuredOutput`, emits one `structured-output.complete` event |
|
|
193
|
+
|
|
194
|
+
**Native combined mode vs fallback** is signaled by the adapter's
|
|
195
|
+
optional `supportsCombinedToolsAndSchema(modelOptions)` method. When
|
|
196
|
+
it returns `true`, the engine wires the JSON Schema into the regular
|
|
197
|
+
`chatStream` call and harvests the final-turn text — middleware sees
|
|
198
|
+
the run through `beforeModel` / `modelStream` as usual, and the
|
|
199
|
+
`'structuredOutput'` middleware phase does **not** fire. When it
|
|
200
|
+
returns `false` (or is omitted), the engine takes the legacy
|
|
201
|
+
finalization path: agent loop, then a separate `structuredOutput` /
|
|
202
|
+
`structuredOutputStream` call with `'structuredOutput'` phase tagging.
|
|
203
|
+
|
|
204
|
+
Consumer code is identical across providers — always read the final object off `structured-output.complete`.
|
|
193
205
|
|
|
194
206
|
### Pattern 4: useChat with outputSchema (progressive UI)
|
|
195
207
|
|
|
@@ -123,6 +123,29 @@ export interface TextAdapter<
|
|
|
123
123
|
structuredOutputStream?: (
|
|
124
124
|
options: StructuredOutputOptions<TProviderOptions>,
|
|
125
125
|
) => AsyncIterable<StreamChunk>
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Declares whether the adapter supports combining `tools` and a
|
|
129
|
+
* schema-constrained final answer in a single streaming request.
|
|
130
|
+
*
|
|
131
|
+
* When `true`, the engine wires `outputSchema` into the regular
|
|
132
|
+
* `chatStream()` call and skips the separate `runStructuredFinalization`
|
|
133
|
+
* round-trip. The model's natural final turn carries the
|
|
134
|
+
* schema-constrained JSON text and the engine harvests it from the agent
|
|
135
|
+
* loop's accumulated content.
|
|
136
|
+
*
|
|
137
|
+
* When `false`, `undefined`, or the method is omitted, the engine runs
|
|
138
|
+
* the agent loop without `outputSchema` and then issues a separate
|
|
139
|
+
* `structuredOutput` / `structuredOutputStream` call against the JSON
|
|
140
|
+
* schema for finalization (the legacy path).
|
|
141
|
+
*
|
|
142
|
+
* The method receives the per-call `modelOptions` so providers whose
|
|
143
|
+
* support depends on the resolved upstream model (e.g. OpenRouter) can
|
|
144
|
+
* answer per-request. Most adapters can return a constant.
|
|
145
|
+
*/
|
|
146
|
+
supportsCombinedToolsAndSchema?: (
|
|
147
|
+
modelOptions?: TProviderOptions | undefined,
|
|
148
|
+
) => boolean
|
|
126
149
|
}
|
|
127
150
|
|
|
128
151
|
/**
|
|
@@ -312,11 +312,20 @@ interface TextEngineConfig<
|
|
|
312
312
|
* as the validated result and retrievable via
|
|
313
313
|
* `getValidatedStructuredOutput()`. Used by `runAgenticStructuredOutput`
|
|
314
314
|
* to perform Standard Schema validation inside the engine.
|
|
315
|
+
* - nativeCombined: when true, the adapter declared
|
|
316
|
+
* `supportsCombinedToolsAndSchema()` and the engine wires `jsonSchema`
|
|
317
|
+
* into the regular `chatStream` call instead of running a separate
|
|
318
|
+
* finalization round-trip. The agent loop's final-turn text is the
|
|
319
|
+
* schema-constrained JSON; the engine parses it from accumulated
|
|
320
|
+
* content. The `'structuredOutput'` middleware phase does NOT fire on
|
|
321
|
+
* this path — middleware sees the run through `beforeModel` /
|
|
322
|
+
* `modelStream` as usual.
|
|
315
323
|
*/
|
|
316
324
|
finalStructuredOutput?: {
|
|
317
325
|
jsonSchema: JSONSchema
|
|
318
326
|
yieldChunks: boolean
|
|
319
327
|
validate?: (data: unknown) => unknown
|
|
328
|
+
nativeCombined?: boolean
|
|
320
329
|
}
|
|
321
330
|
}
|
|
322
331
|
|
|
@@ -379,6 +388,16 @@ class TextEngine<
|
|
|
379
388
|
// Structured-output finalization state (populated by runStructuredFinalization)
|
|
380
389
|
private structuredOutputResult: { data: unknown; rawText: string } | null =
|
|
381
390
|
null
|
|
391
|
+
// Native combined mode: tracks whether we've already emitted the synthetic
|
|
392
|
+
// `structured-output.start` event before the schema-constrained final-turn
|
|
393
|
+
// text begins streaming. The event must precede the first
|
|
394
|
+
// TEXT_MESSAGE_START so the client-side StreamProcessor routes the JSON
|
|
395
|
+
// deltas into a StructuredOutputPart instead of a plain TextPart.
|
|
396
|
+
private combinedStartEmitted = false
|
|
397
|
+
// Native combined mode: messageId we want the synthetic
|
|
398
|
+
// `structured-output.start` (and any error emitted before deltas arrive)
|
|
399
|
+
// to carry, so the client matches it to the streaming text deltas.
|
|
400
|
+
private combinedStructuredMessageId: string | null = null
|
|
382
401
|
// Holds the validated value when `finalStructuredOutput.validate` is provided
|
|
383
402
|
// and succeeds. Distinct from `structuredOutputResult.data` (the raw,
|
|
384
403
|
// unvalidated payload from the structured-output.complete chunk).
|
|
@@ -393,6 +412,7 @@ class TextEngine<
|
|
|
393
412
|
jsonSchema: JSONSchema
|
|
394
413
|
yieldChunks: boolean
|
|
395
414
|
validate?: (data: unknown) => unknown
|
|
415
|
+
nativeCombined?: boolean
|
|
396
416
|
}
|
|
397
417
|
|
|
398
418
|
constructor(
|
|
@@ -560,12 +580,19 @@ class TextEngine<
|
|
|
560
580
|
return
|
|
561
581
|
}
|
|
562
582
|
|
|
563
|
-
// Skip the agent loop entirely when there are no tools AND a
|
|
564
|
-
// output finalization will run. Without tools the model has
|
|
565
|
-
// do in the loop, so executing one iteration would burn an
|
|
566
|
-
// provider call before the finalization request.
|
|
583
|
+
// Skip the agent loop entirely when there are no tools AND a separate
|
|
584
|
+
// structured-output finalization will run. Without tools the model has
|
|
585
|
+
// nothing to do in the loop, so executing one iteration would burn an
|
|
586
|
+
// extra provider call before the finalization request.
|
|
587
|
+
//
|
|
588
|
+
// Native combined mode does NOT skip — the agent loop itself produces
|
|
589
|
+
// the schema-constrained final answer in one pass (model emits the
|
|
590
|
+
// schema-constrained text on its natural final turn). Even with zero
|
|
591
|
+
// tools, the single chatStream call IS the structured-output call.
|
|
567
592
|
const skipAgentLoop =
|
|
568
|
-
!!this.finalStructuredOutput &&
|
|
593
|
+
!!this.finalStructuredOutput &&
|
|
594
|
+
this.tools.length === 0 &&
|
|
595
|
+
this.finalStructuredOutput.nativeCombined !== true
|
|
569
596
|
|
|
570
597
|
if (!skipAgentLoop) {
|
|
571
598
|
do {
|
|
@@ -584,11 +611,12 @@ class TextEngine<
|
|
|
584
611
|
this.middlewareCtx.phase = 'beforeModel'
|
|
585
612
|
this.middlewareCtx.iteration = this.iterationCount
|
|
586
613
|
const iterConfig = this.buildMiddlewareConfig()
|
|
587
|
-
const
|
|
588
|
-
this.
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
614
|
+
const iterTransformedConfig =
|
|
615
|
+
await this.middlewareRunner.runOnConfig(
|
|
616
|
+
this.middlewareCtx,
|
|
617
|
+
iterConfig,
|
|
618
|
+
)
|
|
619
|
+
this.applyMiddlewareConfig(iterTransformedConfig)
|
|
592
620
|
|
|
593
621
|
yield* this.streamModelResponse()
|
|
594
622
|
} else {
|
|
@@ -607,12 +635,20 @@ class TextEngine<
|
|
|
607
635
|
// requested AND the run hasn't already errored/aborted, run it through
|
|
608
636
|
// the middleware pipeline. The terminal hook fires once at the very
|
|
609
637
|
// end (after finalization), not after the agent loop.
|
|
638
|
+
//
|
|
639
|
+
// Native combined mode takes a different path: the agent loop's final-
|
|
640
|
+
// turn text IS the schema-constrained JSON, so we harvest it from
|
|
641
|
+
// `accumulatedContent` instead of issuing a second provider call.
|
|
610
642
|
if (
|
|
611
643
|
this.finalStructuredOutput &&
|
|
612
644
|
!this.isCancelled() &&
|
|
613
645
|
!this.finalizationError
|
|
614
646
|
) {
|
|
615
|
-
|
|
647
|
+
if (this.finalStructuredOutput.nativeCombined === true) {
|
|
648
|
+
yield* this.harvestCombinedStructuredOutput()
|
|
649
|
+
} else {
|
|
650
|
+
yield* this.runStructuredFinalization()
|
|
651
|
+
}
|
|
616
652
|
}
|
|
617
653
|
|
|
618
654
|
// Call terminal hook (skip when waiting for client — stream is paused, not finished).
|
|
@@ -777,6 +813,18 @@ class TextEngine<
|
|
|
777
813
|
},
|
|
778
814
|
)
|
|
779
815
|
|
|
816
|
+
// When the adapter declared `supportsCombinedToolsAndSchema()`, the
|
|
817
|
+
// activity layer set `nativeCombined: true` and we forward the
|
|
818
|
+
// pre-converted JSON Schema into the regular chatStream call. The
|
|
819
|
+
// adapter wires it into the upstream request (e.g. `response_format`,
|
|
820
|
+
// `text.format`, `output_format`) so the model's final-turn text is
|
|
821
|
+
// schema-constrained and the engine can harvest it from the agent loop
|
|
822
|
+
// without a separate finalization round-trip.
|
|
823
|
+
const combinedSchema =
|
|
824
|
+
this.finalStructuredOutput?.nativeCombined === true
|
|
825
|
+
? this.finalStructuredOutput.jsonSchema
|
|
826
|
+
: undefined
|
|
827
|
+
|
|
780
828
|
for await (const chunk of this.adapter.chatStream({
|
|
781
829
|
model: this.params.model,
|
|
782
830
|
messages: this.messages,
|
|
@@ -792,6 +840,7 @@ class TextEngine<
|
|
|
792
840
|
threadId: this.threadId,
|
|
793
841
|
runId: this.runIdOverride,
|
|
794
842
|
parentRunId: this.parentRunIdOverride,
|
|
843
|
+
...(combinedSchema ? { outputSchema: combinedSchema } : {}),
|
|
795
844
|
})) {
|
|
796
845
|
if (this.isCancelled()) {
|
|
797
846
|
break
|
|
@@ -803,6 +852,44 @@ class TextEngine<
|
|
|
803
852
|
// BEFORE middleware, so fields like finishReason, delta, etc. are available
|
|
804
853
|
this.handleStreamChunk(chunk)
|
|
805
854
|
|
|
855
|
+
// Native combined mode: synthesize `structured-output.start` BEFORE
|
|
856
|
+
// the first TEXT_MESSAGE_START so the client-side StreamProcessor
|
|
857
|
+
// routes the schema-constrained JSON deltas into a
|
|
858
|
+
// StructuredOutputPart. We delay synthesis until we actually see
|
|
859
|
+
// text starting — intermediate tool-call iterations don't need it,
|
|
860
|
+
// and emitting at run-start would wrap tool-call commentary into a
|
|
861
|
+
// structured-output part too.
|
|
862
|
+
if (
|
|
863
|
+
this.finalStructuredOutput?.nativeCombined === true &&
|
|
864
|
+
this.finalStructuredOutput.yieldChunks &&
|
|
865
|
+
!this.combinedStartEmitted &&
|
|
866
|
+
chunk.type === EventType.TEXT_MESSAGE_START
|
|
867
|
+
) {
|
|
868
|
+
this.combinedStartEmitted = true
|
|
869
|
+
const messageId =
|
|
870
|
+
typeof chunk.messageId === 'string' && chunk.messageId !== ''
|
|
871
|
+
? chunk.messageId
|
|
872
|
+
: generateMessageId()
|
|
873
|
+
this.combinedStructuredMessageId = messageId
|
|
874
|
+
const synthStart: StreamChunk = {
|
|
875
|
+
type: EventType.CUSTOM,
|
|
876
|
+
name: 'structured-output.start',
|
|
877
|
+
value: { messageId },
|
|
878
|
+
model: this.params.model,
|
|
879
|
+
timestamp: Date.now(),
|
|
880
|
+
threadId: this.threadId,
|
|
881
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
882
|
+
}
|
|
883
|
+
const synthOutputs = await this.middlewareRunner.runOnChunk(
|
|
884
|
+
this.middlewareCtx,
|
|
885
|
+
synthStart,
|
|
886
|
+
)
|
|
887
|
+
for (const outputChunk of synthOutputs) {
|
|
888
|
+
yield outputChunk
|
|
889
|
+
this.middlewareCtx.chunkIndex++
|
|
890
|
+
}
|
|
891
|
+
}
|
|
892
|
+
|
|
806
893
|
// Pipe chunk through middleware (devtools middleware observes; strip-to-spec cleans)
|
|
807
894
|
const outputChunks = await this.middlewareRunner.runOnChunk(
|
|
808
895
|
this.middlewareCtx,
|
|
@@ -812,8 +899,13 @@ class TextEngine<
|
|
|
812
899
|
// the agent loop, suppress the agent-loop's RUN_STARTED/RUN_FINISHED
|
|
813
900
|
// here — the finalization step emits the single outer lifecycle pair
|
|
814
901
|
// that reaches the consumer.
|
|
902
|
+
//
|
|
903
|
+
// Native combined mode does NOT issue a second adapter stream — the
|
|
904
|
+
// agent loop's lifecycle IS the outer pair the consumer sees.
|
|
815
905
|
const suppressAgentLifecycle =
|
|
816
|
-
!!this.finalStructuredOutput &&
|
|
906
|
+
!!this.finalStructuredOutput &&
|
|
907
|
+
this.finalStructuredOutput.yieldChunks &&
|
|
908
|
+
this.finalStructuredOutput.nativeCombined !== true
|
|
817
909
|
for (const outputChunk of outputChunks) {
|
|
818
910
|
if (
|
|
819
911
|
suppressAgentLifecycle &&
|
|
@@ -1948,6 +2040,179 @@ class TextEngine<
|
|
|
1948
2040
|
}
|
|
1949
2041
|
}
|
|
1950
2042
|
|
|
2043
|
+
/**
|
|
2044
|
+
* Native combined mode: harvest the structured output from the agent
|
|
2045
|
+
* loop's accumulated final-turn text (no separate provider call).
|
|
2046
|
+
*
|
|
2047
|
+
* The adapter wired `outputSchema` into the regular `chatStream` request,
|
|
2048
|
+
* so the model's final-turn text is the schema-constrained JSON. We parse
|
|
2049
|
+
* `this.accumulatedContent`, populate `this.structuredOutputResult`, emit
|
|
2050
|
+
* a synthetic `structured-output.complete` (and a `structured-output.start`
|
|
2051
|
+
* if one wasn't emitted earlier — only happens on the streaming path when
|
|
2052
|
+
* the model returned no text at all), and run the validate callback when
|
|
2053
|
+
* present. Failures populate `this.finalizationError` so the engine's
|
|
2054
|
+
* terminal-hook chooser routes to `onError` (per spec §7.3).
|
|
2055
|
+
*
|
|
2056
|
+
* The `'structuredOutput'` middleware phase intentionally does NOT fire on
|
|
2057
|
+
* this path — middleware sees the run through `beforeModel` / `modelStream`
|
|
2058
|
+
* as usual. See PR #605 / issue #605 for the design rationale.
|
|
2059
|
+
*/
|
|
2060
|
+
private async *harvestCombinedStructuredOutput(): AsyncGenerator<StreamChunk> {
|
|
2061
|
+
if (!this.finalStructuredOutput) {
|
|
2062
|
+
throw new Error(
|
|
2063
|
+
'harvestCombinedStructuredOutput called without finalStructuredOutput config',
|
|
2064
|
+
)
|
|
2065
|
+
}
|
|
2066
|
+
|
|
2067
|
+
const yieldChunks = this.finalStructuredOutput.yieldChunks
|
|
2068
|
+
const rawText = this.accumulatedContent
|
|
2069
|
+
|
|
2070
|
+
// Empty final-turn text means the agent loop terminated without the
|
|
2071
|
+
// model emitting any assistant content (e.g. early termination after
|
|
2072
|
+
// tool calls). Mirror the fallback path's "missing structured result"
|
|
2073
|
+
// error rather than silently returning undefined.
|
|
2074
|
+
if (rawText.length === 0) {
|
|
2075
|
+
this.finalizationError = {
|
|
2076
|
+
message: 'missing structured result',
|
|
2077
|
+
code: 'structured-output-missing-result',
|
|
2078
|
+
}
|
|
2079
|
+
} else {
|
|
2080
|
+
try {
|
|
2081
|
+
const parsed: unknown = JSON.parse(rawText)
|
|
2082
|
+
this.structuredOutputResult = { data: parsed, rawText }
|
|
2083
|
+
} catch (err: unknown) {
|
|
2084
|
+
const detail =
|
|
2085
|
+
rawText.slice(0, 200) + (rawText.length > 200 ? '...' : '')
|
|
2086
|
+
this.finalizationError = {
|
|
2087
|
+
message: `Failed to parse structured output as JSON. Content: ${detail}`,
|
|
2088
|
+
code: 'structured-output-parse-failed',
|
|
2089
|
+
cause: err,
|
|
2090
|
+
}
|
|
2091
|
+
}
|
|
2092
|
+
}
|
|
2093
|
+
|
|
2094
|
+
// Validate against the Standard Schema (when supplied). Validation
|
|
2095
|
+
// failures route through onError just like the fallback path.
|
|
2096
|
+
if (
|
|
2097
|
+
this.structuredOutputResult &&
|
|
2098
|
+
!this.finalizationError &&
|
|
2099
|
+
this.finalStructuredOutput.validate
|
|
2100
|
+
) {
|
|
2101
|
+
try {
|
|
2102
|
+
const validated = this.finalStructuredOutput.validate(
|
|
2103
|
+
this.structuredOutputResult.data,
|
|
2104
|
+
)
|
|
2105
|
+
this.validatedStructuredOutput = validated
|
|
2106
|
+
this.hasValidatedStructuredOutput = true
|
|
2107
|
+
} catch (err: unknown) {
|
|
2108
|
+
const message = err instanceof Error ? err.message : String(err)
|
|
2109
|
+
this.finalizationError = {
|
|
2110
|
+
message,
|
|
2111
|
+
code: 'structured-output-validation-failed',
|
|
2112
|
+
cause: err,
|
|
2113
|
+
}
|
|
2114
|
+
}
|
|
2115
|
+
}
|
|
2116
|
+
|
|
2117
|
+
if (!yieldChunks) {
|
|
2118
|
+
// Promise<T> path: state is populated, nothing to yield. The
|
|
2119
|
+
// activity-layer caller pulls `structuredOutputResult` /
|
|
2120
|
+
// `validatedStructuredOutput` directly.
|
|
2121
|
+
return
|
|
2122
|
+
}
|
|
2123
|
+
|
|
2124
|
+
// Streaming path: emit a synthetic `structured-output.start` if the
|
|
2125
|
+
// model produced no text at all (so the client snaps an errored
|
|
2126
|
+
// StructuredOutputPart rather than nothing). The normal path already
|
|
2127
|
+
// emitted start before the first TEXT_MESSAGE_START in
|
|
2128
|
+
// `streamModelResponse`.
|
|
2129
|
+
if (!this.combinedStartEmitted) {
|
|
2130
|
+
this.combinedStartEmitted = true
|
|
2131
|
+
const messageId = this.combinedStructuredMessageId ?? generateMessageId()
|
|
2132
|
+
this.combinedStructuredMessageId = messageId
|
|
2133
|
+
const synthStart: StreamChunk = {
|
|
2134
|
+
type: EventType.CUSTOM,
|
|
2135
|
+
name: 'structured-output.start',
|
|
2136
|
+
value: { messageId },
|
|
2137
|
+
model: this.params.model,
|
|
2138
|
+
timestamp: Date.now(),
|
|
2139
|
+
threadId: this.threadId,
|
|
2140
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
2141
|
+
}
|
|
2142
|
+
const startOutputs = await this.middlewareRunner.runOnChunk(
|
|
2143
|
+
this.middlewareCtx,
|
|
2144
|
+
synthStart,
|
|
2145
|
+
)
|
|
2146
|
+
for (const outputChunk of startOutputs) {
|
|
2147
|
+
yield outputChunk
|
|
2148
|
+
this.middlewareCtx.chunkIndex++
|
|
2149
|
+
}
|
|
2150
|
+
}
|
|
2151
|
+
|
|
2152
|
+
// On success, emit the synthetic `structured-output.complete` carrying
|
|
2153
|
+
// the parsed object + raw text. Pin the messageId so the client-side
|
|
2154
|
+
// handler can target the right UIMessage even when the agent loop's
|
|
2155
|
+
// terminal RUN_FINISHED has already cleared `activeMessageIds` (the
|
|
2156
|
+
// complete event yields AFTER the loop ends, by which point
|
|
2157
|
+
// `getActiveAssistantMessageId()` returns null and would otherwise drop
|
|
2158
|
+
// the event silently).
|
|
2159
|
+
if (this.structuredOutputResult && !this.finalizationError) {
|
|
2160
|
+
const completeChunk: StreamChunk = {
|
|
2161
|
+
type: EventType.CUSTOM,
|
|
2162
|
+
name: 'structured-output.complete',
|
|
2163
|
+
value: {
|
|
2164
|
+
object: this.structuredOutputResult.data,
|
|
2165
|
+
raw: this.structuredOutputResult.rawText,
|
|
2166
|
+
...(this.combinedStructuredMessageId
|
|
2167
|
+
? { messageId: this.combinedStructuredMessageId }
|
|
2168
|
+
: {}),
|
|
2169
|
+
},
|
|
2170
|
+
model: this.params.model,
|
|
2171
|
+
timestamp: Date.now(),
|
|
2172
|
+
threadId: this.threadId,
|
|
2173
|
+
...(this.runIdOverride ? { runId: this.runIdOverride } : {}),
|
|
2174
|
+
}
|
|
2175
|
+
const completeOutputs = await this.middlewareRunner.runOnChunk(
|
|
2176
|
+
this.middlewareCtx,
|
|
2177
|
+
completeChunk,
|
|
2178
|
+
)
|
|
2179
|
+
for (const outputChunk of completeOutputs) {
|
|
2180
|
+
yield outputChunk
|
|
2181
|
+
this.middlewareCtx.chunkIndex++
|
|
2182
|
+
}
|
|
2183
|
+
}
|
|
2184
|
+
|
|
2185
|
+
// On failure, emit a synthetic RUN_ERROR so the streaming consumer's
|
|
2186
|
+
// `for await` doesn't end silently. Mirrors the fallback path.
|
|
2187
|
+
if (this.finalizationError) {
|
|
2188
|
+
const errChunk: StreamChunk = {
|
|
2189
|
+
type: EventType.RUN_ERROR,
|
|
2190
|
+
runId: this.runIdOverride ?? this.requestId,
|
|
2191
|
+
model: this.params.model,
|
|
2192
|
+
timestamp: Date.now(),
|
|
2193
|
+
threadId: this.threadId,
|
|
2194
|
+
message: this.finalizationError.message,
|
|
2195
|
+
...(this.finalizationError.code
|
|
2196
|
+
? { code: this.finalizationError.code }
|
|
2197
|
+
: {}),
|
|
2198
|
+
error: {
|
|
2199
|
+
message: this.finalizationError.message,
|
|
2200
|
+
...(this.finalizationError.code
|
|
2201
|
+
? { code: this.finalizationError.code }
|
|
2202
|
+
: {}),
|
|
2203
|
+
},
|
|
2204
|
+
}
|
|
2205
|
+
const errOutputs = await this.middlewareRunner.runOnChunk(
|
|
2206
|
+
this.middlewareCtx,
|
|
2207
|
+
errChunk,
|
|
2208
|
+
)
|
|
2209
|
+
for (const outputChunk of errOutputs) {
|
|
2210
|
+
yield outputChunk
|
|
2211
|
+
this.middlewareCtx.chunkIndex++
|
|
2212
|
+
}
|
|
2213
|
+
}
|
|
2214
|
+
}
|
|
2215
|
+
|
|
1951
2216
|
private buildMiddlewareConfig(): ChatMiddlewareConfig {
|
|
1952
2217
|
return {
|
|
1953
2218
|
messages: this.messages,
|
|
@@ -2243,6 +2508,13 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
|
|
|
2243
2508
|
parseWithStandardSchema<InferSchemaType<TSchema>>(outputSchema, data)
|
|
2244
2509
|
: undefined
|
|
2245
2510
|
|
|
2511
|
+
// Per issue #605: same capability check as the streaming path. When the
|
|
2512
|
+
// adapter handles tools + schema natively, the engine skips the separate
|
|
2513
|
+
// structured-output finalization call and harvests the JSON from the
|
|
2514
|
+
// agent loop's accumulated final-turn text.
|
|
2515
|
+
const nativeCombined =
|
|
2516
|
+
adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
|
|
2517
|
+
|
|
2246
2518
|
const engine = new TextEngine(
|
|
2247
2519
|
{
|
|
2248
2520
|
adapter,
|
|
@@ -2256,6 +2528,7 @@ async function runAgenticStructuredOutput<TSchema extends SchemaInput>(
|
|
|
2256
2528
|
jsonSchema,
|
|
2257
2529
|
yieldChunks: false,
|
|
2258
2530
|
...(validate ? { validate } : {}),
|
|
2531
|
+
...(nativeCombined ? { nativeCombined: true } : {}),
|
|
2259
2532
|
},
|
|
2260
2533
|
},
|
|
2261
2534
|
logger,
|
|
@@ -2493,6 +2766,16 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
|
|
|
2493
2766
|
const model = adapter.model
|
|
2494
2767
|
const logger = resolveDebugOption(debug)
|
|
2495
2768
|
|
|
2769
|
+
// Per issue #605: adapters that natively combine tools + schema-constrained
|
|
2770
|
+
// output in one streaming call (modern OpenAI, Anthropic 4.5+, Gemini 3+,
|
|
2771
|
+
// Grok 4+) opt in via `supportsCombinedToolsAndSchema()`. The engine then
|
|
2772
|
+
// forwards the schema into the regular `chatStream` call and harvests the
|
|
2773
|
+
// structured result from the agent loop's accumulated text — no separate
|
|
2774
|
+
// finalization round-trip, and the `'structuredOutput'` middleware phase
|
|
2775
|
+
// does not fire.
|
|
2776
|
+
const nativeCombined =
|
|
2777
|
+
adapter.supportsCombinedToolsAndSchema?.(options.modelOptions) === true
|
|
2778
|
+
|
|
2496
2779
|
// Inputs may be UIMessages (from useChat) or ModelMessages (from server-side
|
|
2497
2780
|
// callers). TextEngine handles the conversion uniformly.
|
|
2498
2781
|
const engine = new TextEngine(
|
|
@@ -2504,7 +2787,11 @@ async function* runStreamingStructuredOutputImpl<TSchema extends SchemaInput>(
|
|
|
2504
2787
|
>,
|
|
2505
2788
|
middleware,
|
|
2506
2789
|
context,
|
|
2507
|
-
finalStructuredOutput: {
|
|
2790
|
+
finalStructuredOutput: {
|
|
2791
|
+
jsonSchema,
|
|
2792
|
+
yieldChunks: true,
|
|
2793
|
+
...(nativeCombined ? { nativeCombined: true } : {}),
|
|
2794
|
+
},
|
|
2508
2795
|
},
|
|
2509
2796
|
logger,
|
|
2510
2797
|
)
|